@aria-framework/ai 0.23.0 → 0.24.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -1
- package/index.js +31 -1
- package/package.json +1 -1
- package/providers/lmx.js +0 -0
- package/providers/openai-compatible.js +8 -0
- package/views/ai/lmx-stack.ejs +11 -6
package/README.md
CHANGED
|
@@ -134,6 +134,14 @@ for (const step of rekeyPlan(rows, doc.engines)) {
|
|
|
134
134
|
over to the id. A row whose id has disappeared is left alone, and the panel shows it as
|
|
135
135
|
missing. It is never re-pointed at whichever engine now has its old name.
|
|
136
136
|
|
|
137
|
+
To find a stored row's engine in a status document, use the same rule the router, the panel and
|
|
138
|
+
the verifier use. Import it from the package root, not from `providers/lmxDiscovery`:
|
|
139
|
+
|
|
140
|
+
```js
|
|
141
|
+
const { findEngine } = require('@aria-framework/ai');
|
|
142
|
+
const engine = findEngine(doc.engines, { id: row.lmx_engine_id || null, name: row.lmx_engine });
|
|
143
|
+
```
|
|
144
|
+
|
|
137
145
|
### Errors and routing
|
|
138
146
|
|
|
139
147
|
When an lmx call cannot run, the thrown `AiError` carries `err.lmxSkip`. None of these mean the
|
|
@@ -152,7 +160,13 @@ endpoints in preference order and take the first one that serves.
|
|
|
152
160
|
|
|
153
161
|
### Embeddings
|
|
154
162
|
|
|
155
|
-
`embed(
|
|
163
|
+
Call `client.embed(cfg, texts, { signal })` and pass your own resolved embedding config; it never
|
|
164
|
+
falls back to `resolveConfig()`. It applies the same retry as `complete()`: a `429` with
|
|
165
|
+
`Retry-After` is waited out once, and a fast dropped connection is retried once. Calling the
|
|
166
|
+
provider adapter's `embed()` directly skips both, and a throttled batch then fails outright.
|
|
167
|
+
Embeddings are not metered against the token budget.
|
|
168
|
+
|
|
169
|
+
On lmx, `embed()` puts the engine's `dimensions` on the config as `embeddingDimensions`. Record the width
|
|
156
170
|
you built an index with and compare it on startup. A model swap changes the width, and an index
|
|
157
171
|
built at the old width becomes quietly wrong rather than loudly broken.
|
|
158
172
|
|
package/index.js
CHANGED
|
@@ -145,6 +145,32 @@ function createAiClient(deps = {}) {
|
|
|
145
145
|
return result;
|
|
146
146
|
}
|
|
147
147
|
|
|
148
|
+
/**
|
|
149
|
+
* Embed one or more strings, with the same retry policy as complete().
|
|
150
|
+
*
|
|
151
|
+
* THE RETRY IS THE POINT. complete() always went through withOneRetry; embeddings were called on
|
|
152
|
+
* the adapter directly, so a gateway 429 failed the whole batch even though Retry-After said how
|
|
153
|
+
* long to wait - the README promised otherwise, and lmx's CLIENT.md requires a wait-and-retry for
|
|
154
|
+
* every request through the gateway. Embeddings are idempotent, so a resend is always safe.
|
|
155
|
+
*
|
|
156
|
+
* THE CONFIG IS THE CALLER'S. An app resolves its embedding settings separately from completion
|
|
157
|
+
* (embeddingModel, often a different endpoint), so this never falls back to resolveConfig().
|
|
158
|
+
* No budget metering - embeddings were never metered, and starting is a separate decision.
|
|
159
|
+
* @param {object} cfg the resolved embedding config (provider, embeddingModel, …)
|
|
160
|
+
* @param {string|string[]} texts
|
|
161
|
+
* @param {{signal?: AbortSignal}} [opts]
|
|
162
|
+
*/
|
|
163
|
+
async function embed(cfg, texts, opts = {}) {
|
|
164
|
+
const adapter = cfg && PROVIDERS[cfg.provider];
|
|
165
|
+
if (!adapter) {
|
|
166
|
+
throw new AiError('unconfigured', `No adapter is registered for provider "${cfg && cfg.provider}".`);
|
|
167
|
+
}
|
|
168
|
+
if (typeof adapter.embed !== 'function') {
|
|
169
|
+
throw new AiError('unsupported', `${cfg.label || cfg.provider} does not support embeddings.`);
|
|
170
|
+
}
|
|
171
|
+
return withOneRetry(() => adapter.embed(cfg, texts), opts);
|
|
172
|
+
}
|
|
173
|
+
|
|
148
174
|
/** Is there a provider configured at all? Callers use this to decide whether to render a control. */
|
|
149
175
|
async function isEnabled() {
|
|
150
176
|
return (await resolveConfig()).enabled;
|
|
@@ -217,7 +243,7 @@ function createAiClient(deps = {}) {
|
|
|
217
243
|
const boundGenerate = (opts) => generate(complete, opts);
|
|
218
244
|
|
|
219
245
|
return {
|
|
220
|
-
complete, isEnabled, test, listModels, listModelsResult, withOneRetry,
|
|
246
|
+
complete, embed, isEnabled, test, listModels, listModelsResult, withOneRetry,
|
|
221
247
|
benchmark: benchmarkEndpoint,
|
|
222
248
|
polish: boundPolish, generate: boundGenerate,
|
|
223
249
|
facts, AiError, PROVIDERS, DEFAULTS
|
|
@@ -260,6 +286,10 @@ module.exports = {
|
|
|
260
286
|
// Does this stack actually work? Four checks in the only order they can run. No database and no
|
|
261
287
|
// keystore — every credential arrives as an argument — so it loads eagerly.
|
|
262
288
|
lmxVerify: require('./lmxVerify'),
|
|
289
|
+
// Engine identity - (instance, id), falling back to name - shared by the router, the stack panel
|
|
290
|
+
// and the verifier. Apps used to deep-require providers/lmxDiscovery for it; that path still
|
|
291
|
+
// works, but this is the supported one.
|
|
292
|
+
findEngine: require('./providers/lmxDiscovery').findEngine,
|
|
263
293
|
// The counting rules a screen needs. Separate from the verifier because "what did the stack say"
|
|
264
294
|
// and "what does that mean for what I am relying on" are different questions, and only the second
|
|
265
295
|
// one needs to know which engines this app has adopted.
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@aria-framework/ai",
|
|
3
3
|
"description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.24.1",
|
|
5
5
|
"license": "UNLICENSED",
|
|
6
6
|
"private": false,
|
|
7
7
|
"publishConfig": {
|
package/providers/lmx.js
CHANGED
|
Binary file
|
|
@@ -404,6 +404,14 @@ async function embed(cfg, texts) {
|
|
|
404
404
|
} finally {
|
|
405
405
|
clearTimeout(timer);
|
|
406
406
|
}
|
|
407
|
+
// THE STATUS FIRST, the body second — the order complete() already had. Parsing first turned a
|
|
408
|
+
// 429 (often an empty body) into "not JSON", and a 429 with a JSON body into "could not embed",
|
|
409
|
+
// both `bad_response`: never `rate_limit`, so client.embed's Retry-After wait (0.24.0) could not
|
|
410
|
+
// fire and an lmx gateway throttle read as a broken engine. httpError is handed the body already
|
|
411
|
+
// read — a real Response cannot be read twice.
|
|
412
|
+
if (res.status === 401 || res.status === 403 || res.status === 429) {
|
|
413
|
+
throw await httpError({ status: res.status, headers: res.headers, text: async () => raw }, label, cfg.apiKey);
|
|
414
|
+
}
|
|
407
415
|
let payload;
|
|
408
416
|
try { payload = JSON.parse(raw); } catch (err) {
|
|
409
417
|
throw new AiError('bad_response', `${label} returned something that is not JSON from ${url}.`);
|
package/views/ai/lmx-stack.ejs
CHANGED
|
@@ -84,8 +84,9 @@
|
|
|
84
84
|
<%# WHAT EDIT PREFILLS. The certificate IS carried, because it is public — it is not even
|
|
85
85
|
encrypted at rest — and an operator opening Edit to a blank certificate box has no way to
|
|
86
86
|
tell a stored one from none at all. The two tokens are carried only as LENGTHS, which is
|
|
87
|
-
the most that can be said about a write-only secret and
|
|
88
|
-
|
|
87
|
+
the most that can be said about a write-only secret, and enough for an operator to see
|
|
88
|
+
whether the two match: they SHOULD through the gateway (one lmxk_ key in both fields) and
|
|
89
|
+
should NOT direct to lmx (status.token and an engine key are different credentials). %>
|
|
89
90
|
data-lmx-cert="<%= inst.caCert || '' %>"
|
|
90
91
|
data-lmx-token-len="<%= (inst.creds && inst.creds.statusTokenLength) || '' %>"
|
|
91
92
|
data-lmx-key-len="<%= (inst.creds && inst.creds.enginesKeyLength) || '' %>"
|
|
@@ -367,21 +368,25 @@
|
|
|
367
368
|
</div>
|
|
368
369
|
|
|
369
370
|
<%# THE LENGTHS ARE THE POINT. Both are write-only, so the only thing that can be said about
|
|
370
|
-
a stored secret is how long it is
|
|
371
|
-
|
|
371
|
+
a stored secret is how long it is. Equal lengths are expected through the gateway, where
|
|
372
|
+
one lmxk_ key is both credentials; direct to lmx they are two different secrets, and
|
|
373
|
+
equal lengths there are worth a second look. The help text says which is which. %>
|
|
372
374
|
<div class="col-md-6">
|
|
373
375
|
<label class="form-label" for="lmx-token-<%= inst.id %>">Status token</label>
|
|
374
376
|
<input id="lmx-token-<%= inst.id %>" type="password" class="form-control" name="status_token"
|
|
375
377
|
autocomplete="new-password"
|
|
376
378
|
placeholder="<%= (inst.creds && inst.creds.statusTokenLength) ? 'unchanged — ' + inst.creds.statusTokenLength + ' characters stored' : 'nothing stored — enter a value' %>">
|
|
377
|
-
<div class="form-text">Reads the status document
|
|
379
|
+
<div class="form-text">Reads the status document; never sent to an engine. Direct to lmx: the
|
|
380
|
+
<code>status.token</code>. Through the gateway: your <code>lmxk_</code> key, the same value
|
|
381
|
+
as the engines key.</div>
|
|
378
382
|
</div>
|
|
379
383
|
<div class="col-md-6">
|
|
380
384
|
<label class="form-label" for="lmx-key-<%= inst.id %>">Engines key</label>
|
|
381
385
|
<input id="lmx-key-<%= inst.id %>" type="password" class="form-control" name="engines_key"
|
|
382
386
|
autocomplete="new-password"
|
|
383
387
|
placeholder="<%= (inst.creds && inst.creds.enginesKeyLength) ? 'unchanged — ' + inst.creds.enginesKeyLength + ' characters stored' : 'nothing stored — enter a value' %>">
|
|
384
|
-
<div class="form-text">Used by every engine here unless one overrides it
|
|
388
|
+
<div class="form-text">Used by every engine here unless one overrides it. Through the gateway:
|
|
389
|
+
the same <code>lmxk_</code> key as the status token.</div>
|
|
385
390
|
</div>
|
|
386
391
|
|
|
387
392
|
<div class="col-12">
|