@aria-framework/ai 0.21.0 → 0.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +179 -0
- package/index.js +17 -2
- package/lmxStatus.js +61 -10
- package/lmxStore.js +8 -2
- package/lmxVerify.js +4 -2
- package/package.json +4 -3
- package/polish.js +21 -1
- package/providerStore.js +8 -1
- package/providers/lmx.js +0 -0
- package/providers/lmxDiscovery.js +31 -10
- package/providers/openai-compatible.js +6 -1
- package/views/ai/lmx-stack.ejs +1 -0
package/README.md
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
# @aria-framework/ai
|
|
2
|
+
|
|
3
|
+
A model seam shared by apps: one client (`createAiClient`) over several providers, plus the
|
|
4
|
+
writing engines (Polish, Generate), a fact-preservation guard, an untrusted-text fence, and
|
|
5
|
+
the stores and views behind an AI admin page. Prompts, routing policy and credentials stay in
|
|
6
|
+
the app that uses the package.
|
|
7
|
+
|
|
8
|
+
| Provider | `provider` value | Address |
|
|
9
|
+
|---|---|---|
|
|
10
|
+
| LM Studio | `lmstudio` | `baseUrl` (default `http://localhost:1234/v1`) |
|
|
11
|
+
| Any OpenAI-compatible server | `openai-compatible` | `baseUrl` |
|
|
12
|
+
| Anthropic | `anthropic` | `https://api.anthropic.com/v1` |
|
|
13
|
+
| **lmx supervised stack** | `lmx` | **discovered at run time**, never configured |
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
npm install @aria-framework/ai undici # undici only if you use lmx with a pinned certificate
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## The client
|
|
20
|
+
|
|
21
|
+
```js
|
|
22
|
+
const { createAiClient } = require('@aria-framework/ai');
|
|
23
|
+
|
|
24
|
+
const ai = createAiClient({
|
|
25
|
+
// Called on every request, so a settings change needs no restart.
|
|
26
|
+
resolveConfig: async () => ({
|
|
27
|
+
enabled: true, provider: 'openai-compatible',
|
|
28
|
+
baseUrl: 'http://gpu-box:8080/v1', model: 'qwen3', apiKey: '...',
|
|
29
|
+
timeoutMs: 60000, maxTokens: 1024, contextTokens: 8192
|
|
30
|
+
}),
|
|
31
|
+
budget: { async assertWithinBudget(cfg, meta) {}, async record(cfg, result, meta) {} }, // optional
|
|
32
|
+
logger: console, // optional
|
|
33
|
+
maxRetryAfterMs: 10000 // optional: the longest Retry-After a call will wait out (see 429 below)
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
const r = await ai.complete({ system, messages, maxTokens: 400, schema /* optional */ });
|
|
37
|
+
// r.text, r.json (when schema), r.model, r.usage.total, r.ms
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Every failure is thrown as an `AiError` with a `kind`: `disabled`, `unconfigured`, `unreachable`,
|
|
41
|
+
`timeout`, `auth`, `rate_limit`, `bad_response` or `refused`. The `message` is written for
|
|
42
|
+
an admin screen and never includes a URL or key.
|
|
43
|
+
|
|
44
|
+
`providerStore`, `usageStore`, `speedStore` and `lmxStore` are optional db-worker-backed
|
|
45
|
+
stores. Each exports a `schemaFor(dialect)`, so the app's migration can be checked against it.
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## lmx
|
|
50
|
+
|
|
51
|
+
lmx is a supervisor that runs llama.cpp engines and publishes a **status document** saying
|
|
52
|
+
which engines exist, which model each one runs and what state it is in. The full contract is
|
|
53
|
+
`CLIENT.md` in the lmx repo, and this adapter implements it. The rules you would otherwise have
|
|
54
|
+
to write yourself:
|
|
55
|
+
|
|
56
|
+
| Contract rule | What the adapter does |
|
|
57
|
+
|---|---|
|
|
58
|
+
| Poll `/status` about every 2 s | One poller per `(instance, statusUrl)`, shared by every engine on that stack, with `unref()` called on it |
|
|
59
|
+
| Route only to `healthy`; `draining` gets nothing new | Selects **for** `healthy`, so a state lmx adds later can never receive work by accident |
|
|
60
|
+
| Engine URLs come from the document | Resolved on every call, and never stored |
|
|
61
|
+
| A status outage is not an inference outage | Keeps routing on the last good document for 3 minutes (`staleMs`), logging loudly |
|
|
62
|
+
| Pin the certificate; never disable verification | Uses an undici `Agent({ connect: { ca } })` per stack. Never `NODE_EXTRA_CA_CERTS`, never `rejectUnauthorized:false` |
|
|
63
|
+
| Choose the reasoning flag from the model | `qwen` → `chat_template_kwargs.enable_thinking=false`, `gpt-oss` → `reasoning_effort:'low'`, anything else gets neither (with a warning) |
|
|
64
|
+
| Size requests against `maxInputTokens` (per slot) | `contextTokens` is taken from the engine, not from config |
|
|
65
|
+
| Identity is `(instance, id)`, falling back to `name` | `lmx.engineId` is matched first; `rekeyPlan()` migrates rows when ids are minted or engines renamed |
|
|
66
|
+
| A `429` is about the key, not the engine | Waits `Retry-After` once (up to `maxRetryAfterMs`), then throws with `lmxSkip: 'lmx_throttled'` |
|
|
67
|
+
| The document's `instance` must match | A document for another instance is refused, because engine names collide across stacks |
|
|
68
|
+
|
|
69
|
+
### Ask the operator for
|
|
70
|
+
|
|
71
|
+
- **Direct to lmx:** the status URL (`https://<host>:9443/status`), `lmx.crt`, `status.token`,
|
|
72
|
+
and an engine key (`lmx keygen --add`). **Never** take `admin.token`.
|
|
73
|
+
- **Through the gateway:** the status URL `https://<host>/e/status`, the certificate if it is
|
|
74
|
+
self-signed, and **one** `infer`-scoped `lmxk_…` key. That one key is both the status token
|
|
75
|
+
and the engines key, so put the same value in both fields.
|
|
76
|
+
|
|
77
|
+
The adapter takes the status URL exactly as given; it does not append `/status`.
|
|
78
|
+
|
|
79
|
+
### Config shape
|
|
80
|
+
|
|
81
|
+
`resolveConfig()` (or the `cfgOverride` you pass to `complete`) returns:
|
|
82
|
+
|
|
83
|
+
```js
|
|
84
|
+
{
|
|
85
|
+
enabled: true,
|
|
86
|
+
provider: 'lmx',
|
|
87
|
+
apiKey: '<engines key>', // sent to the ENGINE as Bearer
|
|
88
|
+
timeoutMs: 60000, maxTokens: 1024,
|
|
89
|
+
logger, // optional, gets the "status unreachable" warnings
|
|
90
|
+
lmx: {
|
|
91
|
+
instance: 'site-a-dev', // MUST equal the document's top-level `instance`
|
|
92
|
+
statusUrl: 'https://lmx-host:9443/status', // or https://gateway/e/status
|
|
93
|
+
statusToken: '<status token>', // the same lmxk_ key as apiKey through the gateway
|
|
94
|
+
ca: '-----BEGIN CERTIFICATE-----…', // PEM to pin; null = system trust store
|
|
95
|
+
engine: 'advanced', // engine name: always stored, used for display and logs
|
|
96
|
+
engineId: '7c9e6679-…', // engine id when the stack publishes one; null before mint-ids
|
|
97
|
+
pollMs: 2000, staleMs: 180000 // optional
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
`model` is not needed. The engine reports which model it runs, and anything you send is
|
|
103
|
+
ignored.
|
|
104
|
+
|
|
105
|
+
### What to store
|
|
106
|
+
|
|
107
|
+
The operator assigns engines to jobs, and the supervisor never does. For each choice, store:
|
|
108
|
+
|
|
109
|
+
- `instance`, which is the stack's own name, checked against the document.
|
|
110
|
+
- `engine_id`, when present. It is the only field that survives a rename.
|
|
111
|
+
- `engine` (the name), as the fallback before `mint-ids` and for display.
|
|
112
|
+
|
|
113
|
+
Do **not** store the URL, port, model, or `label`. The label is for display only and should be
|
|
114
|
+
re-read on every poll.
|
|
115
|
+
|
|
116
|
+
`providerStore.schemaFor()` has `lmx_instance`, `lmx_engine` and `lmx_engine_id`.
|
|
117
|
+
`lmxStore.schemaFor()` holds the supervisor row (`id` = instance, `status_url`, poll and stale
|
|
118
|
+
intervals). Credentials belong in your own secret store; this package never sees one at rest.
|
|
119
|
+
|
|
120
|
+
### Re-keying when ids appear or engines are renamed
|
|
121
|
+
|
|
122
|
+
Call this after each status read, for example from the same refresh that updates labels:
|
|
123
|
+
|
|
124
|
+
```js
|
|
125
|
+
const { rekeyPlan } = require('@aria-framework/ai');
|
|
126
|
+
for (const step of rekeyPlan(rows, doc.engines)) {
|
|
127
|
+
// { id: 'row-id', set: { lmx_engine_id: 'uuid' }, why: 'minted' }
|
|
128
|
+
// { id: 'row-id', set: { lmx_engine: 'new-name' }, why: 'renamed', from: 'old-name' }
|
|
129
|
+
await updateRow(step.id, step.set); // and audit it
|
|
130
|
+
}
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
`minted` means the *same* engine has gained an id, so its routes, health and history carry
|
|
134
|
+
over to the id. A row whose id has disappeared is left alone, and the panel shows it as
|
|
135
|
+
missing. It is never re-pointed at whichever engine now has its old name.
|
|
136
|
+
|
|
137
|
+
### Errors and routing
|
|
138
|
+
|
|
139
|
+
When an lmx call cannot run, the thrown `AiError` carries `err.lmxSkip`. None of these mean the
|
|
140
|
+
engine is broken, so **do not trip a circuit breaker on them**. Move to the next endpoint
|
|
141
|
+
instead:
|
|
142
|
+
|
|
143
|
+
| `lmxSkip` | Meaning |
|
|
144
|
+
|---|---|
|
|
145
|
+
| `lmx_not_healthy` | draining, restarting or starting; the supervisor is doing planned work |
|
|
146
|
+
| `lmx_stale` | no fresh status document, so we cannot see the stack |
|
|
147
|
+
| `lmx_unknown_engine` | the engine is not in the document (removed, or hidden from this gateway key) |
|
|
148
|
+
| `lmx_throttled` | a `429` on the key; the one `Retry-After` wait has already been spent |
|
|
149
|
+
|
|
150
|
+
Failover across several engines for one job is the app's job; lmx provides none. List
|
|
151
|
+
endpoints in preference order and take the first one that serves.
|
|
152
|
+
|
|
153
|
+
### Embeddings
|
|
154
|
+
|
|
155
|
+
`embed()` puts the engine's `dimensions` on the config as `embeddingDimensions`. Record the width
|
|
156
|
+
you built an index with and compare it on startup. A model swap changes the width, and an index
|
|
157
|
+
built at the old width becomes quietly wrong rather than loudly broken.
|
|
158
|
+
|
|
159
|
+
### Operator screens
|
|
160
|
+
|
|
161
|
+
- `lmxVerify.verify({ instance, statusUrl, statusToken, caCert, enginesKey, engineId?, engineName? })`
|
|
162
|
+
runs the ordered checks (reachable → certificate → status token → engines key). The
|
|
163
|
+
engines-key check uses `/v1/models`, so it costs no tokens and works through the gateway.
|
|
164
|
+
- `describeStack()` and `views/ai/lmx-stack.ejs` draw the stack panel (active, draining, missing
|
|
165
|
+
and available engines, and certificate expiry).
|
|
166
|
+
- `listModelsResult(cfg)` returns the engines for the picker, with `id`, `label`, `role`, model
|
|
167
|
+
facts and the reasoning flag.
|
|
168
|
+
|
|
169
|
+
### Untrusted content
|
|
170
|
+
|
|
171
|
+
A ticket, an email or a log line is attacker-controlled text. Pass it through `untrusted.js`
|
|
172
|
+
or `fenced.js`, which supply delimited, provenance-tagged data. Use a `schema` so decoding is
|
|
173
|
+
constrained. Verify extracted entities against your own records, and keep a person in the loop
|
|
174
|
+
for anything that changes state.
|
|
175
|
+
|
|
176
|
+
### Conformance
|
|
177
|
+
|
|
178
|
+
Run lmx's `examples/client-smoke.mjs` against the stack before blaming this package. Exit
|
|
179
|
+
code 0 means the contract holds, so a failure is on this side.
|
package/index.js
CHANGED
|
@@ -50,6 +50,8 @@ const DEFAULTS = {
|
|
|
50
50
|
|
|
51
51
|
const RETRY_AFTER_MS = 400;
|
|
52
52
|
const RETRY_ONLY_IF_FAILED_WITHIN_MS = 5000;
|
|
53
|
+
/** The longest Retry-After a call will sit out before giving up on this endpoint. */
|
|
54
|
+
const MAX_RETRY_AFTER_MS = 10000;
|
|
53
55
|
|
|
54
56
|
const NOOP_LOGGER = { info() {}, warn() {}, error() {} };
|
|
55
57
|
const NOOP_BUDGET = { async assertWithinBudget() {}, async record() {} };
|
|
@@ -65,11 +67,13 @@ function createAiClient(deps = {}) {
|
|
|
65
67
|
}
|
|
66
68
|
const log = deps.logger || NOOP_LOGGER;
|
|
67
69
|
const meter = deps.budget || NOOP_BUDGET;
|
|
70
|
+
const maxRetryAfterMs = Number.isFinite(deps.maxRetryAfterMs) ? deps.maxRetryAfterMs : MAX_RETRY_AFTER_MS;
|
|
68
71
|
|
|
69
72
|
/**
|
|
70
73
|
* Try once more, but only for the failures where trying again could help — a local provider that
|
|
71
|
-
* dropped the connection while loading a model,
|
|
72
|
-
* rate limit or a slow failure is never retried (see the
|
|
74
|
+
* dropped the connection while loading a model, or a rate limit that said how long to wait. A
|
|
75
|
+
* cancelled call, a timeout, an open-ended rate limit or a slow failure is never retried (see the
|
|
76
|
+
* guards below).
|
|
73
77
|
*/
|
|
74
78
|
async function withOneRetry(run, opts = {}) {
|
|
75
79
|
const startedAt = Date.now();
|
|
@@ -78,6 +82,17 @@ function createAiClient(deps = {}) {
|
|
|
78
82
|
} catch (err) {
|
|
79
83
|
if (opts.signal && opts.signal.aborted) throw err;
|
|
80
84
|
if (!err || !err.retryable) throw err;
|
|
85
|
+
// A 429 THAT SAYS HOW LONG. The lmx gateway's per-key limits answer with Retry-After, and its
|
|
86
|
+
// contract is explicit: the limit is on the KEY, so wait and retry rather than blaming the
|
|
87
|
+
// engine. Bounded, because a request a person is waiting on cannot sit out a minute; past the
|
|
88
|
+
// cap it throws as before and the caller's routing decides.
|
|
89
|
+
if (err.kind === 'rate_limit' && Number.isFinite(err.retryAfterMs)
|
|
90
|
+
&& err.retryAfterMs <= maxRetryAfterMs) {
|
|
91
|
+
log.warn(`AI: rate limited — waiting ${err.retryAfterMs}ms as asked, then trying once more`);
|
|
92
|
+
await new Promise((r) => setTimeout(r, err.retryAfterMs));
|
|
93
|
+
if (opts.signal && opts.signal.aborted) throw err;
|
|
94
|
+
return run();
|
|
95
|
+
}
|
|
81
96
|
const elapsed = Date.now() - startedAt;
|
|
82
97
|
// A TIMEOUT is the deadline itself being reached — retrying waits the whole deadline again. A
|
|
83
98
|
// RATE LIMIT is the provider asking for less pressure. A slow `unreachable` is not the
|
package/lmxStatus.js
CHANGED
|
@@ -32,6 +32,7 @@
|
|
|
32
32
|
'use strict';
|
|
33
33
|
|
|
34
34
|
const { reasoningFor } = require('./providers/lmx');
|
|
35
|
+
const { findEngine } = require('./providers/lmxDiscovery');
|
|
35
36
|
|
|
36
37
|
/** Inside this window a certificate is worth mentioning: an expired pin fails at the handshake. */
|
|
37
38
|
const EXPIRY_WARN_DAYS = 30;
|
|
@@ -88,17 +89,22 @@ function describeStack(o = {}) {
|
|
|
88
89
|
// NULL, NOT [] — "we never got a document" and "this stack has no engines" send an operator to
|
|
89
90
|
// completely different places, and only the first one means "press Check".
|
|
90
91
|
const reported = (o.report && o.report.engines) ? o.report.engines.map(describeEngine) : null;
|
|
91
|
-
const byName = new Map((reported || []).map((e) => [e.name, e]));
|
|
92
92
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
93
|
+
// THE SAME MATCH THE ROUTER MAKES — by id when the row has one, by name otherwise. A panel that
|
|
94
|
+
// matched on name alone would call a renamed engine "missing" while calls to it went on working.
|
|
95
|
+
const added = rows.map((row) => {
|
|
96
|
+
const engine = findEngine(reported, rowRef(row));
|
|
97
|
+
return {
|
|
98
|
+
row,
|
|
99
|
+
engine,
|
|
100
|
+
missing: !!reported && !engine,
|
|
101
|
+
health: health[row.id] || null,
|
|
102
|
+
routes: routes[row.id] || []
|
|
103
|
+
};
|
|
104
|
+
});
|
|
100
105
|
|
|
101
|
-
const
|
|
106
|
+
const adopted = new Set(added.map((a) => a.engine).filter(Boolean));
|
|
107
|
+
const available = (reported || []).filter((e) => !adopted.has(e));
|
|
102
108
|
|
|
103
109
|
const activeCount = added
|
|
104
110
|
.filter((a) => Number(a.row.enabled) && a.engine && a.engine.state === 'healthy').length;
|
|
@@ -122,4 +128,49 @@ function describeStack(o = {}) {
|
|
|
122
128
|
};
|
|
123
129
|
}
|
|
124
130
|
|
|
125
|
-
|
|
131
|
+
/** An endpoint row's reference to its engine, in the shape findEngine takes. */
|
|
132
|
+
const rowRef = (row) => ({ id: row.lmx_engine_id || null, name: row.lmx_engine });
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* What to write to the endpoint rows so they follow the stack — the RE-KEY CLIENT.md asks for.
|
|
136
|
+
*
|
|
137
|
+
* Two cases, and only two:
|
|
138
|
+
*
|
|
139
|
+
* minted — the row has no id, and the engine of the same name now carries one. This is
|
|
140
|
+
* `lmx mint-ids` having run on the host. It is the SAME engine, and the row's state
|
|
141
|
+
* (routes, health, history) must carry forward onto the id. Reading it as one engine
|
|
142
|
+
* disappearing and another appearing is the natural mistake, and it loses all of that.
|
|
143
|
+
* renamed — the row has an id, the engine with that id now has a different name. The row keeps
|
|
144
|
+
* its id and takes the new name, so logs and admin URLs say what the stack says.
|
|
145
|
+
*
|
|
146
|
+
* Nothing else moves. A row whose id is no longer reported stays as it is (the panel calls it
|
|
147
|
+
* missing); it is NEVER re-pointed at whatever now carries its old name, because ids are add-only
|
|
148
|
+
* upstream and that is a different engine.
|
|
149
|
+
*
|
|
150
|
+
* Pure: returns a plan, writes nothing. The app owns its rows, its audit trail and its transaction.
|
|
151
|
+
*
|
|
152
|
+
* @param {Array} rows endpoint rows: { id, lmx_engine, lmx_engine_id? }
|
|
153
|
+
* @param {Array|null} engines engines from a status document; null when none arrived
|
|
154
|
+
* @returns {Array<{id, set, why, from?}>}
|
|
155
|
+
*/
|
|
156
|
+
function rekeyPlan(rows, engines) {
|
|
157
|
+
if (!Array.isArray(engines)) return [];
|
|
158
|
+
const plan = [];
|
|
159
|
+
for (const row of rows || []) {
|
|
160
|
+
if (!row) continue;
|
|
161
|
+
if (!row.lmx_engine_id) {
|
|
162
|
+
const e = findEngine(engines, { name: row.lmx_engine });
|
|
163
|
+
if (e && e.id) plan.push({ id: row.id, set: { lmx_engine_id: e.id }, why: 'minted' });
|
|
164
|
+
continue;
|
|
165
|
+
}
|
|
166
|
+
const e = findEngine(engines, { id: row.lmx_engine_id });
|
|
167
|
+
if (e && e.name && e.name !== row.lmx_engine) {
|
|
168
|
+
plan.push({ id: row.id, set: { lmx_engine: e.name }, why: 'renamed', from: row.lmx_engine });
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
return plan;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
module.exports = {
|
|
175
|
+
describeStack, describeEngine, modelName, reasoningLabel, rekeyPlan, EXPIRY_WARN_DAYS
|
|
176
|
+
};
|
package/lmxStore.js
CHANGED
|
@@ -146,10 +146,16 @@ function createLmxStore(opts = {}) {
|
|
|
146
146
|
return { changes: r.changes };
|
|
147
147
|
},
|
|
148
148
|
|
|
149
|
-
/**
|
|
149
|
+
/**
|
|
150
|
+
* The endpoint rows served by this supervisor — the join that groups a stack on screen.
|
|
151
|
+
*
|
|
152
|
+
* `*`, not a column list: `lmx_engine_id` (ai 0.23.0) must reach describeStack for a renamed
|
|
153
|
+
* engine to be matched, and naming it here would turn an app that has not yet migrated the
|
|
154
|
+
* column into an SQL error on every stack panel. The rows carry no credential to leak.
|
|
155
|
+
*/
|
|
150
156
|
async engines(id) {
|
|
151
157
|
return driver.all(
|
|
152
|
-
`SELECT
|
|
158
|
+
`SELECT * FROM ${providers} `
|
|
153
159
|
+ 'WHERE lmx_instance = ? ORDER BY sort_order, id', [String(id)]);
|
|
154
160
|
}
|
|
155
161
|
};
|
package/lmxVerify.js
CHANGED
|
@@ -155,6 +155,7 @@ const skip = (key, text) => ({ key, status: 'skip', text });
|
|
|
155
155
|
* caCert the certificate to pin, PEM
|
|
156
156
|
* enginesKey bearer for inference — checked only when there is an engine to check it against
|
|
157
157
|
* engineName an adopted engine to authenticate against; omitted, the first healthy one is used
|
|
158
|
+
* engineId its stable id, when the row has one — matched instead of the name (renames)
|
|
158
159
|
* fetchImpl test seam; the pinned transport is used when absent
|
|
159
160
|
*/
|
|
160
161
|
async function verify(o = {}) {
|
|
@@ -311,8 +312,9 @@ async function enginesKeyCheck(o, engines, timeoutMs) {
|
|
|
311
312
|
if (!o.enginesKey) {
|
|
312
313
|
return skip('engines_key', 'no engines key set — engines are being called unauthenticated');
|
|
313
314
|
}
|
|
314
|
-
const
|
|
315
|
-
|
|
315
|
+
const { findEngine } = require('./providers/lmxDiscovery');
|
|
316
|
+
const wanted = (o.engineId || o.engineName)
|
|
317
|
+
? findEngine(engines, { id: o.engineId || null, name: o.engineName })
|
|
316
318
|
: engines.find((e) => e && e.state === 'healthy' && e.url);
|
|
317
319
|
if (!wanted || !wanted.url) {
|
|
318
320
|
return skip('engines_key',
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@aria-framework/ai",
|
|
3
|
-
"description": "Aria App Framework
|
|
4
|
-
"version": "0.
|
|
3
|
+
"description": "Aria App Framework \u2014 AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
|
|
4
|
+
"version": "0.23.0",
|
|
5
5
|
"license": "UNLICENSED",
|
|
6
6
|
"private": false,
|
|
7
7
|
"publishConfig": {
|
|
@@ -47,7 +47,8 @@
|
|
|
47
47
|
}
|
|
48
48
|
},
|
|
49
49
|
"scripts": {
|
|
50
|
-
"test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js"
|
|
50
|
+
"test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/speedStore.js && node test/health.js && node test/listModels.js && node test/benchmark.js && node test/lmxDiscovery.js && node test/lmx.js && node test/lmxVerify.js && node test/lmxStore.js && node test/lmxStatus.js && node test/jobCard.js && node test/untrusted.js && node test/packaging.js && node test/views.js && node test/fenced.js && node test/polish.js",
|
|
51
|
+
"prepublishOnly": "node ../../test/packaging.js ai"
|
|
51
52
|
},
|
|
52
53
|
"devDependencies": {
|
|
53
54
|
"undici": "^8.10.0",
|
package/polish.js
CHANGED
|
@@ -97,7 +97,27 @@ function system(opts) {
|
|
|
97
97
|
return lines.join('\n');
|
|
98
98
|
}
|
|
99
99
|
|
|
100
|
-
|
|
100
|
+
/**
|
|
101
|
+
* Two texts are "the same" when they differ only in SPACING NOISE — not when they differ in LINE
|
|
102
|
+
* STRUCTURE.
|
|
103
|
+
*
|
|
104
|
+
* This was `replace(/\s+/g, ' ')`, which collapses newlines along with spaces, and that made the
|
|
105
|
+
* `unchanged` flag structurally wrong for the one mode whose entire job is line structure. `tidy`
|
|
106
|
+
* is instructed to "break walls of text into paragraphs, turn a sequence of steps into a list" —
|
|
107
|
+
* so a SUCCESSFUL tidy of a single paragraph changes nothing but newlines, and the panel said
|
|
108
|
+
* "The model returned the same text — nothing to change." directly beneath the new paragraph
|
|
109
|
+
* break and the model's own note explaining that it had added one.
|
|
110
|
+
*
|
|
111
|
+
* Two contradictory statements and an Accept button, inviting the reader to discard a correct
|
|
112
|
+
* result. Horizontal whitespace is still noise; a line break is content.
|
|
113
|
+
*/
|
|
114
|
+
function normalise(s) {
|
|
115
|
+
return String(s)
|
|
116
|
+
.replace(/\r\n?/g, '\n') // line-ending flavour is not a change
|
|
117
|
+
.replace(/[ \t]+/g, ' ') // how many spaces is not a change
|
|
118
|
+
.replace(/ *\n/g, '\n') // a trailing space before a break is not a change
|
|
119
|
+
.trim();
|
|
120
|
+
}
|
|
101
121
|
|
|
102
122
|
/**
|
|
103
123
|
* @param {(opts:object, cfg?:object) => Promise<object>} complete the client's complete()
|
package/providerStore.js
CHANGED
|
@@ -52,7 +52,11 @@ const FIELDS = [
|
|
|
52
52
|
//
|
|
53
53
|
// An lmx row has no meaningful `base_url`. The address is discovered from the supervisor on every
|
|
54
54
|
// call, because ports move and a stored URL is the one thing the contract says not to keep.
|
|
55
|
-
'lmx_instance', 'lmx_engine'
|
|
55
|
+
'lmx_instance', 'lmx_engine',
|
|
56
|
+
// THE ENGINE'S STABLE ID, when the stack publishes one (lmx `mint-ids`). Names can be renamed by
|
|
57
|
+
// the operator and have been; the id cannot. NULL on a stack that has not minted, where the name
|
|
58
|
+
// is all there is — see lmxStatus.rekeyPlan for how a row gains one without losing its state.
|
|
59
|
+
'lmx_engine_id'
|
|
56
60
|
];
|
|
57
61
|
|
|
58
62
|
const NUMERIC = new Set(['context_tokens', 'max_tokens', 'timeout_ms', 'daily_token_cap', 'enabled',
|
|
@@ -193,6 +197,7 @@ function createProviderStore(opts = {}) {
|
|
|
193
197
|
// three, having originally compared only the first two and missed exactly this.
|
|
194
198
|
lmxInstance: row.lmx_instance || null,
|
|
195
199
|
lmxEngine: row.lmx_engine || null,
|
|
200
|
+
lmxEngineId: row.lmx_engine_id || null,
|
|
196
201
|
// Carried on the resolved config so a speed test can judge a result without a second read.
|
|
197
202
|
// 0 means no expectation was recorded, which is different from "expected to be slow".
|
|
198
203
|
minTokensPerSec: Number(row.min_tokens_per_sec) || 0
|
|
@@ -235,6 +240,8 @@ function schemaFor(dialect) {
|
|
|
235
240
|
-- address is discovered from the supervisor on every call.
|
|
236
241
|
lmx_instance TEXT,
|
|
237
242
|
lmx_engine TEXT,
|
|
243
|
+
-- The engine's stable id (lmx mint-ids), matched before the name. NULL until the stack has one.
|
|
244
|
+
lmx_engine_id TEXT,
|
|
238
245
|
created_at TEXT NOT NULL DEFAULT (${t.now()})
|
|
239
246
|
`;
|
|
240
247
|
}
|
package/providers/lmx.js
CHANGED
|
Binary file
|
|
@@ -27,10 +27,13 @@
|
|
|
27
27
|
* advertised at the same host the client used to reach the listener, because that is an address
|
|
28
28
|
* known to work from where the client is standing.
|
|
29
29
|
*
|
|
30
|
-
* IDENTITY IS (instance,
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
30
|
+
* IDENTITY IS (instance, id), FALLING BACK TO (instance, name). The `id` is the only engine field
|
|
31
|
+
* guaranteed to survive a rename, and this stack has already renamed engines in production
|
|
32
|
+
* (`triage`/`analysis` became `normal`/`advanced`). A deployment that has not run `lmx mint-ids`
|
|
33
|
+
* publishes no id, and there the name is all there is. Names also collide across deployments —
|
|
34
|
+
* `analysis` exists on every stack — so the document's own `instance` is checked on every poll:
|
|
35
|
+
* pointing a status URL at a different stack is caught here rather than discovered as work
|
|
36
|
+
* quietly going to the wrong machine.
|
|
34
37
|
*
|
|
35
38
|
* NOTHING HERE THROWS INTO A REQUEST. A discovery failure is an ABSENCE OF A ROUTE, which the
|
|
36
39
|
* caller turns into "try the next endpoint", not an error that trips a breaker on an engine that
|
|
@@ -150,23 +153,27 @@ function createLmxDiscovery(opts = {}) {
|
|
|
150
153
|
}
|
|
151
154
|
|
|
152
155
|
/**
|
|
153
|
-
* The engine record for `
|
|
156
|
+
* The engine record for `ref`, whatever state it is in — for a health card, which needs to show
|
|
154
157
|
* "draining" rather than "gone".
|
|
158
|
+
*
|
|
159
|
+
* `ref` is a name (string) or `{ id, name }`. WITH AN ID, ONLY THE ID MATCHES. Ids are add-only
|
|
160
|
+
* upstream, so an engine that carries our old name under a different id (or none) is a different
|
|
161
|
+
* engine, and falling back to the name would quietly send our work to it.
|
|
155
162
|
*/
|
|
156
|
-
const engine = (
|
|
163
|
+
const engine = (ref) => findEngine(engines(), ref);
|
|
157
164
|
|
|
158
165
|
/**
|
|
159
|
-
* Where to send work for `name
|
|
166
|
+
* Where to send work for `ref` (a name, or `{ id, name }`), or null.
|
|
160
167
|
*
|
|
161
168
|
* Returns a REASON alongside, because the caller has to distinguish "this engine is fine and busy
|
|
162
169
|
* being replaced" from "we cannot see the stack" from "there is no such engine" — three different
|
|
163
170
|
* things that all mean "not right now" and only one of which is anybody's fault.
|
|
164
171
|
*/
|
|
165
|
-
function resolve(
|
|
172
|
+
function resolve(ref) {
|
|
166
173
|
if (!doc) return { url: null, reason: 'no_document', detail: lastError };
|
|
167
174
|
if (isStale()) return { url: null, reason: 'stale', detail: `${ageSec()}s old` };
|
|
168
175
|
|
|
169
|
-
const e = engine(
|
|
176
|
+
const e = engine(ref);
|
|
170
177
|
if (!e) return { url: null, reason: 'unknown_engine' };
|
|
171
178
|
|
|
172
179
|
// Select FOR healthy. `draining`, `restarting`, and any state invented after this was written
|
|
@@ -206,4 +213,18 @@ function createLmxDiscovery(opts = {}) {
|
|
|
206
213
|
};
|
|
207
214
|
}
|
|
208
215
|
|
|
209
|
-
|
|
216
|
+
/**
|
|
217
|
+
* Find the engine a stored reference means. Shared with lmxStatus so the panel and the router can
|
|
218
|
+
* never disagree about which engine a row is.
|
|
219
|
+
*
|
|
220
|
+
* @param {Array} list engines from a status document
|
|
221
|
+
* @param {string|{id?: string|null, name?: string}} ref
|
|
222
|
+
*/
|
|
223
|
+
function findEngine(list, ref) {
|
|
224
|
+
const r = (ref && typeof ref === 'object') ? ref : { name: ref };
|
|
225
|
+
const all = Array.isArray(list) ? list : [];
|
|
226
|
+
if (r.id) return all.find((e) => e && e.id === r.id) || null;
|
|
227
|
+
return all.find((e) => e && e.name === r.name) || null;
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
module.exports = { createLmxDiscovery, findEngine, POLL_MS, STALE_MS, HEALTHY };
|
|
@@ -252,7 +252,12 @@ async function httpError(res, label, apiKey) {
|
|
|
252
252
|
return new AiError('auth', `${label} rejected the API key.`, { status: res.status });
|
|
253
253
|
}
|
|
254
254
|
if (res.status === 429) {
|
|
255
|
-
|
|
255
|
+
const err = new AiError('rate_limit', `${label} is rate limiting this app.`, { status: res.status });
|
|
256
|
+
// WHOLE SECONDS, as the lmx gateway sends it. An HTTP-date form, or anything unparseable, leaves
|
|
257
|
+
// it unset — and an unset wait is never retried, which is the safe direction to be wrong in.
|
|
258
|
+
const ra = res.headers && typeof res.headers.get === 'function' ? res.headers.get('retry-after') : null;
|
|
259
|
+
if (ra != null && /^\s*\d+\s*$/.test(String(ra))) err.retryAfterMs = Number(ra) * 1000;
|
|
260
|
+
return err;
|
|
256
261
|
}
|
|
257
262
|
if (res.status === 404) {
|
|
258
263
|
// The single most common local mistake: a model name that is not the one loaded.
|
package/views/ai/lmx-stack.ejs
CHANGED
|
@@ -307,6 +307,7 @@
|
|
|
307
307
|
<form method="POST" action="<%= _base %>/lmx/<%= inst.id %>/adopt">
|
|
308
308
|
<input type="hidden" name="_csrf" value="<%= _csrf() %>">
|
|
309
309
|
<input type="hidden" name="engine" value="<%= e.name %>">
|
|
310
|
+
<% if (e.id) { %><input type="hidden" name="engine_id" value="<%= e.id %>"><% } %>
|
|
310
311
|
<input type="hidden" name="label" value="<%= e.label || e.name %>">
|
|
311
312
|
<input type="hidden" name="role" value="<%= e.role || '' %>">
|
|
312
313
|
<input type="hidden" name="max_input_tokens" value="<%= e.maxInputTokens || '' %>">
|