@aria-framework/ai 0.24.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -29,8 +29,7 @@ const ai = createAiClient({
29
29
  timeoutMs: 60000, maxTokens: 1024, contextTokens: 8192
30
30
  }),
31
31
  budget: { async assertWithinBudget(cfg, meta) {}, async record(cfg, result, meta) {} }, // optional
32
- logger: console, // optional
33
- maxRetryAfterMs: 10000 // optional: the longest Retry-After a call will wait out (see 429 below)
32
+ logger: console // optional
34
33
  });
35
34
 
36
35
  const r = await ai.complete({ system, messages, maxTokens: 400, schema /* optional */ });
@@ -55,7 +54,7 @@ to write yourself:
55
54
 
56
55
  | Contract rule | What the adapter does |
57
56
  |---|---|
58
- | Poll `/status` about every 2 s | One poller per `(instance, statusUrl)`, shared by every engine on that stack, with `unref()` called on it |
57
+ | Poll `/status` about every 2 s | One poller per stack (keyed by `instance`), shared by every engine on it, `unref()`'d. Each read is bounded (`statusTimeoutMs`, default 5 s), concurrent reads share one fetch, and `pollMs` is floored at 500 ms |
59
58
  | Route only to `healthy`; `draining` gets nothing new | Selects **for** `healthy`, so a state lmx adds later can never receive work by accident |
60
59
  | Engine URLs come from the document | Resolved on every call, and never stored |
61
60
  | A status outage is not an inference outage | Keeps routing on the last good document for 3 minutes (`staleMs`), logging loudly |
@@ -63,8 +62,8 @@ to write yourself:
63
62
  | Choose the reasoning flag from the model | `qwen` → `chat_template_kwargs.enable_thinking=false`, `gpt-oss` → `reasoning_effort:'low'`, anything else gets neither (with a warning) |
64
63
  | Size requests against `maxInputTokens` (per slot) | `contextTokens` is taken from the engine, not from config |
65
64
  | Identity is `(instance, id)`, falling back to `name` | `lmx.engineId` is matched first; `rekeyPlan()` migrates rows when ids are minted or engines renamed |
66
- | A `429` is about the key, not the engine | Waits `Retry-After` once (up to `maxRetryAfterMs`), then throws with `lmxSkip: 'lmx_throttled'` |
67
- | The document's `instance` must match | A document for another instance is refused, because engine names collide across stacks |
65
+ | A `429` is about the key, not the engine | Fails fast (since 0.25.0): throws `rate_limit` with `lmxSkip: 'lmx_throttled'` and `retryAfterMs`. The client never waits; whether to wait or move on is the caller's decision |
66
+ | The document's `instance` must match | A document for another instance is refused, because engine names collide across stacks. `lmxVerify` then returns `engines: null`, so nothing downstream (`rekeyPlan`) can act on another stack's list |
68
67
 
69
68
  ### Ask the operator for
70
69
 
@@ -94,7 +93,8 @@ The adapter takes the status URL exactly as given; it does not append `/status`.
94
93
  ca: '-----BEGIN CERTIFICATE-----…', // PEM to pin; null = system trust store
95
94
  engine: 'advanced', // engine name: always stored, used for display and logs
96
95
  engineId: '7c9e6679-…', // engine id when the stack publishes one; null before mint-ids
97
- pollMs: 2000, staleMs: 180000 // optional
96
+ pollMs: 2000, staleMs: 180000, // optional; pollMs is floored at 500, null/undefined = default
97
+ statusTimeoutMs: 5000 // optional; one status read's deadline
98
98
  }
99
99
  }
100
100
  ```
@@ -142,6 +142,14 @@ const { findEngine } = require('@aria-framework/ai');
142
142
  const engine = findEngine(doc.engines, { id: row.lmx_engine_id || null, name: row.lmx_engine });
143
143
  ```
144
144
 
145
+ ### Rotating credentials, and removing a stack
146
+
147
+ The poller takes its token, certificate and intervals from the config you pass on **every** call
148
+ (`discoveryFor` reconfigures the live poller since 0.25.0), so a rotated token or a replaced
149
+ certificate takes effect on the next call — no restart. A changed `statusUrl` replaces the poller.
150
+ When you delete or disable a supervisor, call
151
+ `require('@aria-framework/ai/providers/lmx').forgetDiscovery(instance)` so its poller stops.
152
+
145
153
  ### Errors and routing
146
154
 
147
155
  When an lmx call cannot run, the thrown `AiError` carries `err.lmxSkip`. None of these mean the
@@ -153,7 +161,7 @@ instead:
153
161
  | `lmx_not_healthy` | draining, restarting or starting; the supervisor is doing planned work |
154
162
  | `lmx_stale` | no fresh status document, so we cannot see the stack |
155
163
  | `lmx_unknown_engine` | the engine is not in the document (removed, or hidden from this gateway key) |
156
- | `lmx_throttled` | a `429` on the key; the one `Retry-After` wait has already been spent |
164
+ | `lmx_throttled` | a `429` on the gateway key. Not retried by the client; `err.retryAfterMs` says how long the gateway asked for. Every engine on the stack shares the key, so the next endpoint on the same stack will likely be throttled too |
157
165
 
158
166
  Failover across several engines for one job is the app's job; lmx provides none. List
159
167
  endpoints in preference order and take the first one that serves.
@@ -161,10 +169,9 @@ endpoints in preference order and take the first one that serves.
161
169
  ### Embeddings
162
170
 
163
171
  Call `client.embed(cfg, texts, { signal })` and pass your own resolved embedding config; it never
164
- falls back to `resolveConfig()`. It applies the same retry as `complete()`: a `429` with
165
- `Retry-After` is waited out once, and a fast dropped connection is retried once. Calling the
166
- provider adapter's `embed()` directly skips both, and a throttled batch then fails outright.
167
- Embeddings are not metered against the token budget.
172
+ falls back to `resolveConfig()`. It applies the same policy as `complete()`: a fast dropped
173
+ connection is retried once, a `429` fails fast with `retryAfterMs` on the error, and a config with
174
+ `enabled: false` is refused. Embeddings are not metered against the token budget.
168
175
 
169
176
  On lmx, `embed()` puts the engine's `dimensions` on the config as `embeddingDimensions`. Record the width
170
177
  you built an index with and compare it on startup. A model swap changes the width, and an index
package/index.js CHANGED
@@ -50,8 +50,6 @@ const DEFAULTS = {
50
50
 
51
51
  const RETRY_AFTER_MS = 400;
52
52
  const RETRY_ONLY_IF_FAILED_WITHIN_MS = 5000;
53
- /** The longest Retry-After a call will sit out before giving up on this endpoint. */
54
- const MAX_RETRY_AFTER_MS = 10000;
55
53
 
56
54
  const NOOP_LOGGER = { info() {}, warn() {}, error() {} };
57
55
  const NOOP_BUDGET = { async assertWithinBudget() {}, async record() {} };
@@ -67,13 +65,17 @@ function createAiClient(deps = {}) {
67
65
  }
68
66
  const log = deps.logger || NOOP_LOGGER;
69
67
  const meter = deps.budget || NOOP_BUDGET;
70
- const maxRetryAfterMs = Number.isFinite(deps.maxRetryAfterMs) ? deps.maxRetryAfterMs : MAX_RETRY_AFTER_MS;
71
68
 
72
69
  /**
73
- * Try once more, but only for the failures where trying again could help — a local provider that
74
- * dropped the connection while loading a model, or a rate limit that said how long to wait. A
75
- * cancelled call, a timeout, an open-ended rate limit or a slow failure is never retried (see the
76
- * guards below).
70
+ * Try once more, but only for the failure where trying again could help — a local provider that
71
+ * dropped the connection while loading a model. A cancelled call, a timeout, a rate limit or a slow
72
+ * failure is never retried (see the guards below).
73
+ *
74
+ * A 429 FAILS FAST (0.25.0). 0.23/0.24 slept its Retry-After here (up to 10 s), below every app's
75
+ * dispatcher: every engine on an lmx stack shares one gateway key, so each failover re-hit the same
76
+ * throttle and paid the wait again, and an OpenAI-compatible primary that sent Retry-After delayed a
77
+ * healthy backup by the full wait. Decided 2026-10-01: the client does not wait. The error carries
78
+ * `retryAfterMs` (and `lmxSkip: 'lmx_throttled'` from an lmx engine) for a caller that chooses to.
77
79
  */
78
80
  async function withOneRetry(run, opts = {}) {
79
81
  const startedAt = Date.now();
@@ -82,17 +84,6 @@ function createAiClient(deps = {}) {
82
84
  } catch (err) {
83
85
  if (opts.signal && opts.signal.aborted) throw err;
84
86
  if (!err || !err.retryable) throw err;
85
- // A 429 THAT SAYS HOW LONG. The lmx gateway's per-key limits answer with Retry-After, and its
86
- // contract is explicit: the limit is on the KEY, so wait and retry rather than blaming the
87
- // engine. Bounded, because a request a person is waiting on cannot sit out a minute; past the
88
- // cap it throws as before and the caller's routing decides.
89
- if (err.kind === 'rate_limit' && Number.isFinite(err.retryAfterMs)
90
- && err.retryAfterMs <= maxRetryAfterMs) {
91
- log.warn(`AI: rate limited — waiting ${err.retryAfterMs}ms as asked, then trying once more`);
92
- await new Promise((r) => setTimeout(r, err.retryAfterMs));
93
- if (opts.signal && opts.signal.aborted) throw err;
94
- return run();
95
- }
96
87
  const elapsed = Date.now() - startedAt;
97
88
  // A TIMEOUT is the deadline itself being reached — retrying waits the whole deadline again. A
98
89
  // RATE LIMIT is the provider asking for less pressure. A slow `unreachable` is not the
@@ -146,12 +137,9 @@ function createAiClient(deps = {}) {
146
137
  }
147
138
 
148
139
  /**
149
- * Embed one or more strings, with the same retry policy as complete().
150
- *
151
- * THE RETRY IS THE POINT. complete() always went through withOneRetry; embeddings were called on
152
- * the adapter directly, so a gateway 429 failed the whole batch even though Retry-After said how
153
- * long to wait - the README promised otherwise, and lmx's CLIENT.md requires a wait-and-retry for
154
- * every request through the gateway. Embeddings are idempotent, so a resend is always safe.
140
+ * Embed one or more strings, with the same retry policy as complete() — a fast dropped connection
141
+ * is retried once; a 429 fails fast (0.25.0, see withOneRetry). Embeddings are idempotent, so the
142
+ * one resend is always safe.
155
143
  *
156
144
  * THE CONFIG IS THE CALLER'S. An app resolves its embedding settings separately from completion
157
145
  * (embeddingModel, often a different endpoint), so this never falls back to resolveConfig().
@@ -161,6 +149,11 @@ function createAiClient(deps = {}) {
161
149
  * @param {{signal?: AbortSignal}} [opts]
162
150
  */
163
151
  async function embed(cfg, texts, opts = {}) {
152
+ // EXPLICITLY DISABLED is refused, as complete() refuses it (the review found embed ignored it).
153
+ // Strictly `false`: an embedding config the caller built without the flag is still honoured.
154
+ if (cfg && cfg.enabled === false) {
155
+ throw new AiError('disabled', 'This endpoint is switched off.');
156
+ }
164
157
  const adapter = cfg && PROVIDERS[cfg.provider];
165
158
  if (!adapter) {
166
159
  throw new AiError('unconfigured', `No adapter is registered for provider "${cfg && cfg.provider}".`);
package/lmxStatus.js CHANGED
@@ -155,12 +155,16 @@ const rowRef = (row) => ({ id: row.lmx_engine_id || null, name: row.lmx_engine }
155
155
  */
156
156
  function rekeyPlan(rows, engines) {
157
157
  if (!Array.isArray(engines)) return [];
158
+ // IDS ROWS ALREADY TRACK (0.25.0). CLIENT.md's condition for minting is "an id you have never
159
+ // seen". Without it, renaming row A's engine to the name a stale, id-less row B still carries
160
+ // handed B A's id — B's routes silently went to A's engine.
161
+ const held = new Set((rows || []).map((r) => r && r.lmx_engine_id).filter(Boolean));
158
162
  const plan = [];
159
163
  for (const row of rows || []) {
160
164
  if (!row) continue;
161
165
  if (!row.lmx_engine_id) {
162
166
  const e = findEngine(engines, { name: row.lmx_engine });
163
- if (e && e.id) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
167
+ if (e && e.id && !held.has(e.id)) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
164
168
  continue;
165
169
  }
166
170
  const e = findEngine(engines, { id: row.lmx_engine_id });
package/lmxVerify.js CHANGED
@@ -216,6 +216,11 @@ async function verify(o = {}) {
216
216
  if (doc && doc.instance && String(doc.instance) !== instance) {
217
217
  checks.push(fail('instance',
218
218
  `this stack reports itself as “${doc.instance}”, not “${instance}” — the address points at a different deployment`));
219
+ // AND ITS ENGINES ARE NOT HANDED BACK (0.25.0). The apps' scheduled refresh feeds
220
+ // report.engines to rekeyPlan; returning another deployment's list wrote THAT stack's engine
221
+ // ids onto this stack's rows — irreversibly, since a row with an id is never re-pointed.
222
+ // `null` is "no document for this stack", which every consumer treats as "write nothing".
223
+ found = null;
219
224
  }
220
225
 
221
226
  checks.push(await enginesKeyCheck(o, engines, timeoutMs));
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@aria-framework/ai",
3
3
  "description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
4
- "version": "0.24.0",
4
+ "version": "0.25.0",
5
5
  "license": "UNLICENSED",
6
6
  "private": false,
7
7
  "publishConfig": {
package/providers/lmx.js CHANGED
@@ -27,16 +27,33 @@ const { createLmxDiscovery } = require('./lmxDiscovery');
27
27
  const { lmxTransport } = require('./lmxTransport');
28
28
  const { AiError } = require('../error');
29
29
 
30
- /** Discovery instances, keyed so several engines on one stack share a single poller. */
30
+ /**
31
+ * Discovery instances, ONE PER STACK (keyed by instance), so several engines on one stack share a
32
+ * single poller.
33
+ *
34
+ * Keyed by instance alone since 0.25.0. Keyed by (instance, statusUrl), an edited status URL created
35
+ * a SECOND poller and left the first polling the old address every 2 s for the life of the process.
36
+ */
31
37
  const registry = new Map();
32
38
 
33
- const keyOf = (lmx) => `${lmx.instance}\u0000${lmx.statusUrl}`;
39
+ /** The settings a live poller takes from the app's supervisor row, applied on every use. */
40
+ const settingsOf = (lmx) => ({
41
+ token: lmx.statusToken,
42
+ ca: lmx.ca || null,
43
+ staleMs: lmx.staleMs,
44
+ pollMs: lmx.pollMs,
45
+ fetchTimeoutMs: lmx.statusTimeoutMs
46
+ });
34
47
 
35
48
  /**
36
- * The discovery for this supervisor, created once and started on first use.
49
+ * The discovery for this supervisor, created once and started on first use — and RECONFIGURED on
50
+ * every use (0.25.0).
37
51
  *
38
- * Credentials are NOT part of the key: rotating a token should not orphan a running poller and
39
- * leave the old one polling with a dead credential. The live instance is updated in place instead.
52
+ * Credentials are not part of the key, so rotating a token does not orphan the poller; until 0.25.0
53
+ * nothing then applied the new token either (the comment here said "updated in place"; nothing did),
54
+ * so a rotated key or a replaced certificate never reached the poller and every lmx endpoint went
55
+ * stale until a restart. configure() applies whatever the app hands over each time — the app
56
+ * resolves its supervisor row per call, so the row is the source of truth.
40
57
  */
41
58
  function discoveryFor(lmx, logger) {
42
59
  if (!lmx || !lmx.instance || !lmx.statusUrl) {
@@ -44,29 +61,44 @@ function discoveryFor(lmx, logger) {
44
61
  'This endpoint is an lmx engine but no supervisor is configured for it — an engine name '
45
62
  + 'without a status listener cannot be resolved to an address.');
46
63
  }
47
- const key = keyOf(lmx);
64
+ const key = lmx.instance;
48
65
  let d = registry.get(key);
49
- if (!d) {
50
- d = createLmxDiscovery({
66
+ if (d && d.status().statusUrl !== lmx.statusUrl) {
67
+ // A NEW ADDRESS IS A NEW POLLER: the document, its age and its last error belong to the old one.
68
+ d.stop();
69
+ registry.delete(key);
70
+ d = null;
71
+ }
72
+ if (d) {
73
+ d.configure(settingsOf(lmx));
74
+ } else {
75
+ d = createLmxDiscovery(Object.assign({
51
76
  instance: lmx.instance,
52
77
  statusUrl: lmx.statusUrl,
53
- token: lmx.statusToken,
54
- ca: lmx.ca || null,
55
- staleMs: lmx.staleMs,
56
- // Was accepted from the app's supervisor row and then dropped here, so a configured poll
57
- // interval never took effect. undefined falls through to POLL_MS.
58
- pollMs: lmx.pollMs,
59
78
  // Test seam, the same one createLmxDiscovery already takes: it is the only way to exercise
60
79
  // complete() end-to-end without a live stack, and the cold-start bug lived exactly there.
61
80
  fetchImpl: lmx.fetchImpl,
62
81
  logger
63
- });
82
+ }, settingsOf(lmx)));
64
83
  d.start();
65
84
  registry.set(key, d);
66
85
  }
67
86
  return d;
68
87
  }
69
88
 
89
+ /**
90
+ * Stop and drop the poller for a stack — call it when the app deletes or disables a supervisor
91
+ * (0.25.0). Otherwise that stack's poller kept polling every 2 s for the life of the process.
92
+ * @returns {boolean} whether there was one to forget
93
+ */
94
+ function forgetDiscovery(instance) {
95
+ const d = registry.get(instance);
96
+ if (!d) return false;
97
+ d.stop();
98
+ registry.delete(instance);
99
+ return true;
100
+ }
101
+
70
102
  /**
71
103
  * Why a resolution failed, in the vocabulary the dispatcher classifies on.
72
104
  *
@@ -257,5 +289,5 @@ function _resetRegistry() {
257
289
  module.exports = {
258
290
  complete, embed, listModels, listModelsResult,
259
291
  apiRoot: openai.apiRoot,
260
- reasoningFor, discoveryFor, _resetRegistry, SKIP_CODE
292
+ reasoningFor, discoveryFor, forgetDiscovery, _resetRegistry, SKIP_CODE
261
293
  };
@@ -56,6 +56,25 @@ const STALE_MS = 3 * 60 * 1000;
56
56
  /** The only state that may receive new work. */
57
57
  const HEALTHY = 'healthy';
58
58
 
59
+ /**
60
+ * The fastest a poller may run (0.25.0). A poll interval of 0 — a NULL coerced, a typo — is not a
61
+ * faster poll, it is a busy loop against the supervisor.
62
+ */
63
+ const MIN_POLL_MS = 500;
64
+
65
+ /**
66
+ * How long one status read may take (0.25.0). There was no bound at all — only undici's 300 s
67
+ * defaults — so after a restart an lmx call awaited a hung listener far past its own timeout and
68
+ * the dispatcher never failed over. The document is small and served no-store: seconds is ample.
69
+ */
70
+ const FETCH_TIMEOUT_MS = 5000;
71
+
72
+ // null/undefined mean "the default" (lmxStore stores NULL for exactly that); any number — 0, a
73
+ // typo, NaN — is floored rather than trusted.
74
+ const floorPoll = (ms) => (ms === undefined || ms === null || ms === ''
75
+ ? POLL_MS
76
+ : Math.max(MIN_POLL_MS, Number(ms) || 0));
77
+
59
78
  /**
60
79
  * @param {object} opts
61
80
  * @param {string} opts.statusUrl e.g. https://host:9443/status
@@ -63,7 +82,8 @@ const HEALTHY = 'healthy';
63
82
  * @param {string} opts.token bearer for the status listener
64
83
  * @param {string} [opts.ca] PEM of the certificate to pin
65
84
  * @param {number} [opts.staleMs]
66
- * @param {number} [opts.pollMs]
85
+ * @param {number} [opts.pollMs] floored at MIN_POLL_MS
86
+ * @param {number} [opts.fetchTimeoutMs] one status read's deadline (default FETCH_TIMEOUT_MS)
67
87
  * @param {object} [opts.logger]
68
88
  * @param {function} [opts.fetchImpl] injectable for tests; defaults to global fetch
69
89
  * @param {function} [opts.now] injectable clock
@@ -72,10 +92,6 @@ function createLmxDiscovery(opts = {}) {
72
92
  const {
73
93
  statusUrl,
74
94
  instance,
75
- token,
76
- ca = null,
77
- staleMs = STALE_MS,
78
- pollMs = POLL_MS,
79
95
  logger = console,
80
96
  fetchImpl,
81
97
  now = () => Date.now()
@@ -84,10 +100,21 @@ function createLmxDiscovery(opts = {}) {
84
100
  if (!statusUrl) throw new Error('createLmxDiscovery: statusUrl is required');
85
101
  if (!instance) throw new Error('createLmxDiscovery: instance is required — identity is (instance, name)');
86
102
 
103
+ // MUTABLE, AND THAT IS THE 0.25.0 FIX. These were captured once for the life of the process, so a
104
+ // rotated token or a replaced certificate never reached the poller: Check (a fresh fetch) said
105
+ // "verified" while every poll got 401 and, after staleMs, every engine was skipped as stale until
106
+ // a restart. configure() below updates them in place; the registry calls it on every use.
107
+ let token = opts.token;
108
+ let ca = opts.ca || null;
109
+ let staleMs = opts.staleMs || STALE_MS;
110
+ let pollMs = floorPoll(opts.pollMs);
111
+ let fetchTimeoutMs = Number(opts.fetchTimeoutMs) || FETCH_TIMEOUT_MS;
112
+
87
113
  let doc = null; // the last document that parsed and matched our instance
88
114
  let docAt = 0;
89
115
  let lastError = null;
90
116
  let timer = null;
117
+ let inFlight = null; // the one refresh running now; concurrent callers share it
91
118
 
92
119
  /**
93
120
  * The pinned transport — see providers/lmxTransport.js for why this is not NODE_EXTRA_CA_CERTS
@@ -102,7 +129,9 @@ function createLmxDiscovery(opts = {}) {
102
129
  const t = transport();
103
130
  const res = await t.fetch(statusUrl, {
104
131
  headers: { Authorization: `Bearer ${token}` },
105
- dispatcher: t.dispatcher
132
+ dispatcher: t.dispatcher,
133
+ // BOUNDED (0.25.0): a listener that accepts and never answers is abandoned, not awaited.
134
+ signal: AbortSignal.timeout(fetchTimeoutMs)
106
135
  });
107
136
  if (!res.ok) {
108
137
  const err = new Error(`status listener answered ${res.status}`);
@@ -119,7 +148,17 @@ function createLmxDiscovery(opts = {}) {
119
148
  * failure — the listener answered perfectly — so it is reported as a configuration error, which
120
149
  * is what it is. Adopting it would route `analysis` to another deployment's `analysis`.
121
150
  */
122
- async function refresh() {
151
+ /**
152
+ * SINGLE-FLIGHT (0.25.0). The 2 s poller and every warming AI call each started their own read, so
153
+ * a hung listener stacked a new pending fetch per tick and per request. Concurrent callers now
154
+ * share the one in flight.
155
+ */
156
+ function refresh() {
157
+ if (!inFlight) inFlight = refreshOnce().finally(() => { inFlight = null; });
158
+ return inFlight;
159
+ }
160
+
161
+ async function refreshOnce() {
123
162
  try {
124
163
  const next = await fetchOnce();
125
164
  if (next && next.instance && next.instance !== instance) {
@@ -196,8 +235,28 @@ function createLmxDiscovery(opts = {}) {
196
235
  timer = null;
197
236
  }
198
237
 
238
+ /**
239
+ * Apply new settings to a live poller (0.25.0). `undefined` keeps the current value. A new poll
240
+ * interval restarts a running timer so it takes effect now, not after a process restart.
241
+ * The status URL and instance are identity, not settings: a different URL is a different poller
242
+ * (the registry replaces it — see providers/lmx.js).
243
+ */
244
+ function configure(next = {}) {
245
+ if (next.token !== undefined) token = next.token;
246
+ if (next.ca !== undefined) ca = next.ca || null;
247
+ if (next.staleMs !== undefined) staleMs = next.staleMs || STALE_MS;
248
+ if (next.fetchTimeoutMs !== undefined) fetchTimeoutMs = Number(next.fetchTimeoutMs) || FETCH_TIMEOUT_MS;
249
+ if (next.pollMs !== undefined) {
250
+ const p = floorPoll(next.pollMs);
251
+ if (p !== pollMs) {
252
+ pollMs = p;
253
+ if (timer) { stop(); timer = setInterval(refresh, pollMs); if (timer.unref) timer.unref(); }
254
+ }
255
+ }
256
+ }
257
+
199
258
  return {
200
- start, stop, refresh, resolve, engines, engine,
259
+ start, stop, refresh, resolve, engines, engine, configure,
201
260
  /** The pinned transport, so ENGINE calls reach the same host over the same trust. */
202
261
  transport,
203
262
  /** For a diagnostics panel: what we know and how old it is. */
@@ -208,7 +267,9 @@ function createLmxDiscovery(opts = {}) {
208
267
  engineCount: engines().length,
209
268
  ageSec: doc ? ageSec() : null,
210
269
  stale: doc ? isStale() : true,
211
- lastError
270
+ lastError,
271
+ pollMs,
272
+ running: !!timer
212
273
  })
213
274
  };
214
275
  }
@@ -227,4 +288,4 @@ function findEngine(list, ref) {
227
288
  return all.find((e) => e && e.name === r.name) || null;
228
289
  }
229
290
 
230
- module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY };
291
+ module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY, MIN_POLL_MS, FETCH_TIMEOUT_MS };
@@ -404,6 +404,14 @@ async function embed(cfg, texts) {
404
404
  } finally {
405
405
  clearTimeout(timer);
406
406
  }
407
+ // THE STATUS FIRST, the body second — the order complete() already had. Parsing first turned a
408
+ // 429 (often an empty body) into "not JSON", and a 429 with a JSON body into "could not embed",
409
+ // both `bad_response`: never `rate_limit`, so client.embed's Retry-After wait (0.24.0) could not
410
+ // fire and an lmx gateway throttle read as a broken engine. httpError is handed the body already
411
+ // read — a real Response cannot be read twice.
412
+ if (res.status === 401 || res.status === 403 || res.status === 429) {
413
+ throw await httpError({ status: res.status, headers: res.headers, text: async () => raw }, label, cfg.apiKey);
414
+ }
407
415
  let payload;
408
416
  try { payload = JSON.parse(raw); } catch (err) {
409
417
  throw new AiError('bad_response', `${label} returned something that is not JSON from ${url}.`);