@aria-framework/ai 0.25.0 → 0.26.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.js CHANGED
@@ -1,309 +1,352 @@
1
- /**
2
- * @aria-framework/ai — the AI seam. One `complete()`, several providers behind it, plus the
3
- * writing-assist engines (polish/generate) and the fact-preservation guard.
4
- *
5
- * DEPENDENCY-INJECTED, DATABASE-FREE. The package knows how to talk to a model; it does NOT know
6
- * where an app keeps its settings, its credentials or its token ledger. The consumer builds a
7
- * client with two functions of its own:
8
- *
9
- * const ai = createAiClient({
10
- * resolveConfig, // async () => resolved config (provider, baseUrl, model, apiKey, caps…)
11
- * budget, // { assertWithinBudget(cfg, ctx), record(cfg, result, ctx) } — optional
12
- * logger // { info, warn, error } — optional
13
- * });
14
- *
15
- * The provider adapters already take an explicit config and never read a database, which is what
16
- * makes the seam testable: a stub adapter and a real adapter are called identically.
17
- *
18
- * PROMPTS ARE CONTENT AND LIVE IN THE APP. This package carries the mechanism (how to call a model,
19
- * how to enforce a token ceiling, how to check a rewrite kept its facts) and generic writing
20
- * operations; the words that say "you are editing a reply to a customer" belong to the app.
21
- */
22
-
23
- 'use strict';
24
-
25
- const facts = require('./facts');
26
- const { AiError, fromFetchFailure, redact } = require('./error');
27
- const { polish } = require('./polish');
28
- const { generate } = require('./generate');
29
-
30
- const PROVIDERS = {
31
- // 'lmstudio' and 'openai-compatible' are the SAME adapter with different defaults — a kindness to
32
- // whoever configures it: an operator running LM Studio should not have to know it speaks a shape
33
- // named after somebody else.
34
- lmstudio: require('./providers/openai-compatible'),
35
- 'openai-compatible': require('./providers/openai-compatible'),
36
- anthropic: require('./providers/anthropic'),
37
- // 0.14.0 — a supervised fleet rather than an address. The engine URL is discovered from the
38
- // supervisor's status document per call, the reasoning flag is read from the model the engine
39
- // is actually running, and the whole conversation is pinned to a self-signed certificate.
40
- lmx: require('./providers/lmx')
41
- };
42
-
43
- const DEFAULTS = {
44
- lmstudio: { baseUrl: 'http://localhost:1234/v1', model: 'qwen3.5-9b', label: 'LM Studio' },
45
- 'openai-compatible': { baseUrl: 'http://localhost:11434/v1', model: '', label: 'The model server' },
46
- anthropic: { baseUrl: 'https://api.anthropic.com/v1', model: 'claude-sonnet-4-5', label: 'Claude' },
47
- // No baseUrl: an lmx engine's address is never configured, only discovered.
48
- lmx: { baseUrl: '', model: '', label: 'lmx engine' }
49
- };
50
-
51
- const RETRY_AFTER_MS = 400;
52
- const RETRY_ONLY_IF_FAILED_WITHIN_MS = 5000;
53
-
54
- const NOOP_LOGGER = { info() {}, warn() {}, error() {} };
55
- const NOOP_BUDGET = { async assertWithinBudget() {}, async record() {} };
56
-
57
- /**
58
- * Build an AI client bound to one app's config resolution and token budget.
59
- * @param {{resolveConfig: () => Promise<object>, budget?: object, logger?: object}} deps
60
- */
61
- function createAiClient(deps = {}) {
62
- const resolveConfig = deps.resolveConfig;
63
- if (typeof resolveConfig !== 'function') {
64
- throw new Error('createAiClient: resolveConfig must be an async function returning the resolved config');
65
- }
66
- const log = deps.logger || NOOP_LOGGER;
67
- const meter = deps.budget || NOOP_BUDGET;
68
-
69
- /**
70
- * Try once more, but only for the failure where trying again could help — a local provider that
71
- * dropped the connection while loading a model. A cancelled call, a timeout, a rate limit or a slow
72
- * failure is never retried (see the guards below).
73
- *
74
- * A 429 FAILS FAST (0.25.0). 0.23/0.24 slept its Retry-After here (up to 10 s), below every app's
75
- * dispatcher: every engine on an lmx stack shares one gateway key, so each failover re-hit the same
76
- * throttle and paid the wait again, and an OpenAI-compatible primary that sent Retry-After delayed a
77
- * healthy backup by the full wait. Decided 2026-10-01: the client does not wait. The error carries
78
- * `retryAfterMs` (and `lmxSkip: 'lmx_throttled'` from an lmx engine) for a caller that chooses to.
79
- */
80
- async function withOneRetry(run, opts = {}) {
81
- const startedAt = Date.now();
82
- try {
83
- return await run();
84
- } catch (err) {
85
- if (opts.signal && opts.signal.aborted) throw err;
86
- if (!err || !err.retryable) throw err;
87
- const elapsed = Date.now() - startedAt;
88
- // A TIMEOUT is the deadline itself being reached — retrying waits the whole deadline again. A
89
- // RATE LIMIT is the provider asking for less pressure. A slow `unreachable` is not the
90
- // sub-second dropped-connection transient this retry exists for. None of those retry.
91
- if (err.kind === 'timeout' || err.kind === 'rate_limit' || elapsed > RETRY_ONLY_IF_FAILED_WITHIN_MS) {
92
- log.warn(`AI: ${err.kind} after ${elapsed}ms — not retrying (${err.message})`);
93
- throw err;
94
- }
95
- log.warn(`AI: ${err.kind} — trying once more in ${RETRY_AFTER_MS}ms (${err.message})`);
96
- await new Promise((r) => setTimeout(r, RETRY_AFTER_MS));
97
- return run();
98
- }
99
- }
100
-
101
- /**
102
- * Ask the configured model for something.
103
- * @param {{system?:string, messages:Array, maxTokens?:number, temperature?:number,
104
- * schema?:object, signal?:AbortSignal, ticketId?:*, skipBudget?:boolean}} opts
105
- * @param {object} [cfgOverride] the resolved config, when the caller already has it
106
- */
107
- async function complete(opts, cfgOverride) {
108
- const cfg = cfgOverride || await resolveConfig();
109
- if (!cfg.enabled) {
110
- throw new AiError('disabled', 'AI assistance is switched off. An administrator can enable it in Settings.');
111
- }
112
- // A MODEL NAME IS REQUIRED OF EVERY PROVIDER THAT HAS ONE TO CONFIGURE — which is all of them
113
- // except a supervised stack. There the model is a FACT the supervisor reports about an engine,
114
- // not a setting: the operator picks an engine and whatever it is running answers. Insisting on
115
- // a model name would make an lmx endpoint unusable by demanding the one field the design says
116
- // not to store, and would say so with a message naming nothing an operator could go and fill in.
117
- if (!cfg.model && cfg.provider !== 'lmx') {
118
- throw new AiError('unconfigured', 'No model name is configured.');
119
- }
120
- const adapter = PROVIDERS[cfg.provider];
121
- if (!adapter) {
122
- throw new AiError('unconfigured', `No adapter is registered for provider "${cfg.provider}".`);
123
- }
124
-
125
- // The ceiling, before the call — the only place a limit can be enforced without being bypassable
126
- // by whichever caller forgets. `skipBudget` is for the admin's Test connection; it still records.
127
- if (!opts.skipBudget) await meter.assertWithinBudget(cfg, { ticketId: opts.ticketId });
128
-
129
- const result = await withOneRetry(() => adapter.complete(cfg, opts), opts);
130
-
131
- // ...and the counter after it, AWAITED: two calls in quick succession must both be counted
132
- // before the second's ceiling check reads the total, or the limit is enforced against a stale one.
133
- await meter.record(cfg, result, { ticketId: opts.ticketId });
134
-
135
- log.info(`AI: ${cfg.provider}/${result.model} ${result.usage.total} tokens in ${result.ms}ms`);
136
- return result;
137
- }
138
-
139
- /**
140
- * Embed one or more strings, with the same retry policy as complete() — a fast dropped connection
141
- * is retried once; a 429 fails fast (0.25.0, see withOneRetry). Embeddings are idempotent, so the
142
- * one resend is always safe.
143
- *
144
- * THE CONFIG IS THE CALLER'S. An app resolves its embedding settings separately from completion
145
- * (embeddingModel, often a different endpoint), so this never falls back to resolveConfig().
146
- * No budget metering - embeddings were never metered, and starting is a separate decision.
147
- * @param {object} cfg the resolved embedding config (provider, embeddingModel, …)
148
- * @param {string|string[]} texts
149
- * @param {{signal?: AbortSignal}} [opts]
150
- */
151
- async function embed(cfg, texts, opts = {}) {
152
- // EXPLICITLY DISABLED is refused, as complete() refuses it (the review found embed ignored it).
153
- // Strictly `false`: an embedding config the caller built without the flag is still honoured.
154
- if (cfg && cfg.enabled === false) {
155
- throw new AiError('disabled', 'This endpoint is switched off.');
156
- }
157
- const adapter = cfg && PROVIDERS[cfg.provider];
158
- if (!adapter) {
159
- throw new AiError('unconfigured', `No adapter is registered for provider "${cfg && cfg.provider}".`);
160
- }
161
- if (typeof adapter.embed !== 'function') {
162
- throw new AiError('unsupported', `${cfg.label || cfg.provider} does not support embeddings.`);
163
- }
164
- return withOneRetry(() => adapter.embed(cfg, texts), opts);
165
- }
166
-
167
- /** Is there a provider configured at all? Callers use this to decide whether to render a control. */
168
- async function isEnabled() {
169
- return (await resolveConfig()).enabled;
170
- }
171
-
172
- /** A short round trip for an admin "Test connection". NEVER THROWS — it reports what is wrong. */
173
- async function test(cfgOverride) {
174
- const cfg = cfgOverride || await resolveConfig();
175
- if (!cfg.enabled) return { ok: false, kind: 'disabled', error: 'No provider is selected.' };
176
- try {
177
- const r = await complete({
178
- system: 'Reply with the single word: ready. Do not explain.',
179
- messages: [{ role: 'user', content: 'ready?' }],
180
- maxTokens: 512,
181
- temperature: 0,
182
- skipBudget: true // an admin diagnosing a provider must not be blocked by a full budget
183
- }, cfg);
184
- return {
185
- ok: true, model: r.model, ms: r.ms, reply: (r.text || '').trim().slice(0, 60), usage: r.usage,
186
- finishReason: r.finishReason || null, reasoned: !!r.reasonedFor
187
- };
188
- } catch (err) {
189
- if (err instanceof AiError) return { ok: false, kind: err.kind, error: err.message, retryable: err.retryable };
190
- return { ok: false, kind: 'bad_response', error: err.message };
191
- }
192
- }
193
-
194
- /** What the server has loaded, for an admin model picker. Empty when it cannot say. */
195
- async function listModels(cfgOverride) {
196
- return (await listModelsResult(cfgOverride)).models;
197
- }
198
-
199
- /**
200
- * The model list and WHY it is the length it is.
201
- *
202
- * Callers that only want names should use listModels(). An admin screen wants this one: an empty
203
- * array on its own cannot tell "the server has nothing loaded" from "that address is not an
204
- * OpenAI-compatible API root", and those need different fixes.
205
- */
206
- async function listModelsResult(cfgOverride) {
207
- const cfg = cfgOverride || await resolveConfig();
208
- if (!cfg.enabled) {
209
- return { ok: false, models: [], url: null, status: 0, error: 'No provider is selected.' };
210
- }
211
- const adapter = PROVIDERS[cfg.provider];
212
- if (!adapter) {
213
- return { ok: false, models: [], url: null, status: 0, error: `Unknown provider “${cfg.provider}”.` };
214
- }
215
- try {
216
- if (adapter.listModelsResult) return await adapter.listModelsResult(cfg);
217
- // An adapter that predates this contract still works; it simply cannot explain itself.
218
- return { ok: true, models: await adapter.listModels(cfg), url: null, status: 0, error: null };
219
- } catch (err) {
220
- return { ok: false, models: [], url: null, status: 0, error: err.message };
221
- }
222
- }
223
-
224
- /**
225
- * A fixed-workload speed test for one endpoint. See benchmark.js for why the workload is fixed
226
- * rather than the prompt, and why the warm-up is reported rather than discarded.
227
- */
228
- async function benchmarkEndpoint(cfgOverride, benchOpts) {
229
- const cfg = cfgOverride || await resolveConfig();
230
- return require('./benchmark').benchmark(complete, cfg, benchOpts || {});
231
- }
232
-
233
- // The writing-assist engines are bound to this client's complete() so a caller gets config +
234
- // budget + retry for free. Prompt framing is supplied per call by the app (content).
235
- const boundPolish = (opts) => polish(complete, opts);
236
- const boundGenerate = (opts) => generate(complete, opts);
237
-
238
- return {
239
- complete, embed, isEnabled, test, listModels, listModelsResult, withOneRetry,
240
- benchmark: benchmarkEndpoint,
241
- polish: boundPolish, generate: boundGenerate,
242
- facts, AiError, PROVIDERS, DEFAULTS
243
- };
244
- }
245
-
246
- module.exports = {
247
- // The usage counter behind every ceiling. LAZY: it needs the db-worker driver contract, which
248
- // is an OPTIONAL peer — a consumer using only createAiClient/polish/facts must not be made to
249
- // install a database package to require this one.
250
- get createUsageStore() { return require('./usageStore').createUsageStore; },
251
- get createProviderStore() { return require('./providerStore').createProviderStore; },
252
- // Speed history. Lazy for the same reason as the others: it needs the db-worker driver contract,
253
- // which is an optional peer.
254
- get createSpeedStore() { return require('./speedStore').createSpeedStore; },
255
- // EXPORTED SO A CONSUMER CAN ASSERT ITS TABLE MATCHES. An app writes its own migration, which is
256
- // a hand copy of this DDL — and a copy with nothing comparing it to the original is the failure
257
- // mode this repo has already documented twice. providerSchemaFor and usageSchemaFor exist for the
258
- // same reason; leaving this one out meant a drift would surface as an INSERT throwing at runtime.
259
- get speedSchemaFor() { return require('./speedStore').schemaFor; },
260
- // No database behind health, so it loads eagerly like the rest of the seam.
261
- ...require('./health'),
262
- /**
263
- * Where this package's EJS partials live, for the consumer's view-roots list.
264
- *
265
- * Same contract as backup/server/notify/uploads: the package knows its own layout, the app
266
- * puts its own views FIRST so a local file of the same name wins.
267
- */
268
- viewsDir: require('path').join(__dirname, 'views'),
269
- get providerSchemaFor() { return require('./providerStore').schemaFor; },
270
- // ── SUPERVISED STACKS (lmx) ─────────────────────────────────────────────────────────────────
271
- // The store is LAZY for the same reason as the others: it needs the db-worker driver contract,
272
- // which is an optional peer. A consumer using only createAiClient must not be made to install a
273
- // database package to require this one.
274
- get createLmxStore() { return require('./lmxStore').createLmxStore; },
275
- // EXPORTED SO A CONSUMER CAN ASSERT ITS TABLE MATCHES — the migration is a hand copy of this, and
276
- // a copy with nothing comparing it to the original is the failure this repo has documented three
277
- // times now. The first app to carry the table wrote it with nothing to check against.
278
- get lmxSchemaFor() { return require('./lmxStore').schemaFor; },
279
- // Does this stack actually work? Four checks in the only order they can run. No database and no
280
- // keystore — every credential arrives as an argument — so it loads eagerly.
281
- lmxVerify: require('./lmxVerify'),
282
- // Engine identity - (instance, id), falling back to name - shared by the router, the stack panel
283
- // and the verifier. Apps used to deep-require providers/lmxDiscovery for it; that path still
284
- // works, but this is the supported one.
285
- findEngine: require('./providers/lmxDiscovery').findEngine,
286
- // The counting rules a screen needs. Separate from the verifier because "what did the stack say"
287
- // and "what does that mean for what I am relying on" are different questions, and only the second
288
- // one needs to know which engines this app has adopted.
289
- ...require('./lmxStatus'),
290
- get usageSchemaFor() { return require('./usageStore').schemaFor; },
291
- createAiClient,
292
- PROVIDERS, DEFAULTS,
293
- AiError, fromFetchFailure, redact,
294
- facts,
295
- // THE INPUT SIDE of the same concern facts.js covers on the output side: text somebody else
296
- // wrote, placed where a model can read it without being able to give orders. See
297
- // untrusted.js for why the fence marker has to be generated per call.
298
- untrusted: require('./untrusted'),
299
- // ...and the correct ASSEMBLY of it. untrusted.js hands over `fence()` and `rule()` separately;
300
- // fenced.js puts them together, because one app assembled them two different ways in two prompt
301
- // builders and one of the divergences started refusing correct answers. See fenced.js for the
302
- // measurements, including why a fenced document is worth about a third of the defence unless the
303
- // caller also states its schema's field names.
304
- fenced: require('./fenced'),
305
- // Default writing-op catalogues, so an app can build its menus without re-declaring them.
306
- POLISH_MODES: require('./polish').MODES,
307
- POLISH_TONES: require('./polish').TONES,
308
- RETRY_AFTER_MS
309
- };
1
+ /**
2
+ * @aria-framework/ai — the AI seam. One `complete()`, several providers behind it, plus the
3
+ * writing-assist engines (polish/generate) and the fact-preservation guard.
4
+ *
5
+ * DEPENDENCY-INJECTED, DATABASE-FREE. The package knows how to talk to a model; it does NOT know
6
+ * where an app keeps its settings, its credentials or its token ledger. The consumer builds a
7
+ * client with two functions of its own:
8
+ *
9
+ * const ai = createAiClient({
10
+ * resolveConfig, // async () => resolved config (provider, baseUrl, model, apiKey, caps…)
11
+ * budget, // { assertWithinBudget(cfg, ctx), record(cfg, result, ctx) } — optional
12
+ * logger // { info, warn, error } — optional
13
+ * });
14
+ *
15
+ * The provider adapters already take an explicit config and never read a database, which is what
16
+ * makes the seam testable: a stub adapter and a real adapter are called identically.
17
+ *
18
+ * PROMPTS ARE CONTENT AND LIVE IN THE APP. This package carries the mechanism (how to call a model,
19
+ * how to enforce a token ceiling, how to check a rewrite kept its facts) and generic writing
20
+ * operations; the words that say "you are editing a reply to a customer" belong to the app.
21
+ */
22
+
23
+ 'use strict';
24
+
25
+ const facts = require('./facts');
26
+ const { AiError, fromFetchFailure, redact } = require('./error');
27
+ const { polish } = require('./polish');
28
+ const { generate } = require('./generate');
29
+ const { assertReasoning, REASONING_MODES } = require('./providers/lmx');
30
+
31
+ const PROVIDERS = {
32
+ // 'lmstudio' and 'openai-compatible' are the SAME adapter with different defaults — a kindness to
33
+ // whoever configures it: an operator running LM Studio should not have to know it speaks a shape
34
+ // named after somebody else.
35
+ lmstudio: require('./providers/openai-compatible'),
36
+ 'openai-compatible': require('./providers/openai-compatible'),
37
+ anthropic: require('./providers/anthropic'),
38
+ // 0.14.0 — a supervised fleet rather than an address. The engine URL is discovered from the
39
+ // supervisor's status document per call, the reasoning flag is read from the model the engine
40
+ // is actually running, and the whole conversation is pinned to a self-signed certificate.
41
+ lmx: require('./providers/lmx')
42
+ };
43
+
44
+ const DEFAULTS = {
45
+ lmstudio: { baseUrl: 'http://localhost:1234/v1', model: 'qwen3.5-9b', label: 'LM Studio' },
46
+ 'openai-compatible': { baseUrl: 'http://localhost:11434/v1', model: '', label: 'The model server' },
47
+ anthropic: { baseUrl: 'https://api.anthropic.com/v1', model: 'claude-sonnet-4-5', label: 'Claude' },
48
+ // No baseUrl: an lmx engine's address is never configured, only discovered.
49
+ lmx: { baseUrl: '', model: '', label: 'lmx engine' }
50
+ };
51
+
52
+ const RETRY_AFTER_MS = 400;
53
+ const RETRY_ONLY_IF_FAILED_WITHIN_MS = 5000;
54
+
55
+ /** Providers already warned that they ignore `reasoning` — once per provider per process. */
56
+ const warnedReasoning = new Set();
57
+
58
+ const NOOP_LOGGER = { info() {}, warn() {}, error() {} };
59
+ const NOOP_BUDGET = { async assertWithinBudget() {}, async record() {} };
60
+
61
+ /**
62
+ * Build an AI client bound to one app's config resolution and token budget.
63
+ * @param {{resolveConfig: () => Promise<object>, budget?: object, logger?: object}} deps
64
+ */
65
+ function createAiClient(deps = {}) {
66
+ const resolveConfig = deps.resolveConfig;
67
+ if (typeof resolveConfig !== 'function') {
68
+ throw new Error('createAiClient: resolveConfig must be an async function returning the resolved config');
69
+ }
70
+ const log = deps.logger || NOOP_LOGGER;
71
+ const meter = deps.budget || NOOP_BUDGET;
72
+
73
+ /**
74
+ * Try once more, but only for the failure where trying again could help — a local provider that
75
+ * dropped the connection while loading a model. A cancelled call, a timeout, a rate limit or a slow
76
+ * failure is never retried (see the guards below).
77
+ *
78
+ * A 429 FAILS FAST (0.25.0). 0.23/0.24 slept its Retry-After here (up to 10 s), below every app's
79
+ * dispatcher: every engine on an lmx stack shares one gateway key, so each failover re-hit the same
80
+ * throttle and paid the wait again, and an OpenAI-compatible primary that sent Retry-After delayed a
81
+ * healthy backup by the full wait. Decided 2026-10-01: the client does not wait. The error carries
82
+ * `retryAfterMs` (and `lmxSkip: 'lmx_throttled'` from an lmx engine) for a caller that chooses to.
83
+ */
84
+ async function withOneRetry(run, opts = {}) {
85
+ const startedAt = Date.now();
86
+ try {
87
+ return await run();
88
+ } catch (err) {
89
+ if (opts.signal && opts.signal.aborted) throw err;
90
+ if (!err || !err.retryable) throw err;
91
+ const elapsed = Date.now() - startedAt;
92
+ // A TIMEOUT is the deadline itself being reached — retrying waits the whole deadline again. A
93
+ // RATE LIMIT is the provider asking for less pressure. A slow `unreachable` is not the
94
+ // sub-second dropped-connection transient this retry exists for. None of those retry.
95
+ if (err.kind === 'timeout' || err.kind === 'rate_limit' || elapsed > RETRY_ONLY_IF_FAILED_WITHIN_MS) {
96
+ log.warn(`AI: ${err.kind} after ${elapsed}ms — not retrying (${err.message})`);
97
+ throw err;
98
+ }
99
+ log.warn(`AI: ${err.kind} — trying once more in ${RETRY_AFTER_MS}ms (${err.message})`);
100
+ await new Promise((r) => setTimeout(r, RETRY_AFTER_MS));
101
+ return run();
102
+ }
103
+ }
104
+
105
+ /**
106
+ * Ask the configured model for something.
107
+ * @param {{system?:string, messages:Array, maxTokens?:number, temperature?:number,
108
+ * schema?:object, signal?:AbortSignal, ticketId?:*, skipBudget?:boolean,
109
+ * reasoning?:'full'}} opts
110
+ * @param {object} [cfgOverride] the resolved config, when the caller already has it
111
+ */
112
+ async function complete(opts, cfgOverride) {
113
+ // `reasoning` (0.26.0) is checked here for EVERY provider, first: a typo is the caller's bug
114
+ // whichever endpoint the route happens to pick, and must not wait for an lmx one to surface.
115
+ assertReasoning(opts && opts.reasoning);
116
+ const cfg = cfgOverride || await resolveConfig();
117
+ if (!cfg.enabled) {
118
+ throw new AiError('disabled', 'AI assistance is switched off. An administrator can enable it in Settings.');
119
+ }
120
+ // A MODEL NAME IS REQUIRED OF EVERY PROVIDER THAT HAS ONE TO CONFIGURE — which is all of them
121
+ // except a supervised stack. There the model is a FACT the supervisor reports about an engine,
122
+ // not a setting: the operator picks an engine and whatever it is running answers. Insisting on
123
+ // a model name would make an lmx endpoint unusable by demanding the one field the design says
124
+ // not to store, and would say so with a message naming nothing an operator could go and fill in.
125
+ if (!cfg.model && cfg.provider !== 'lmx') {
126
+ throw new AiError('unconfigured', 'No model name is configured.');
127
+ }
128
+ const adapter = PROVIDERS[cfg.provider];
129
+ if (!adapter) {
130
+ throw new AiError('unconfigured', `No adapter is registered for provider "${cfg.provider}".`);
131
+ }
132
+
133
+ // The ceiling, before the call — the only place a limit can be enforced without being bypassable
134
+ // by whichever caller forgets. `skipBudget` is for the admin's Test connection; it still records.
135
+ if (!opts.skipBudget) await meter.assertWithinBudget(cfg, { ticketId: opts.ticketId });
136
+
137
+ // ONLY lmx HONOURS `reasoning`. Elsewhere it is warned about and removed, so no other adapter's
138
+ // request body can ever carry it.
139
+ let callOpts = opts;
140
+ if (opts.reasoning != null && cfg.provider !== 'lmx') {
141
+ if (!warnedReasoning.has(cfg.provider)) {
142
+ warnedReasoning.add(cfg.provider);
143
+ log.warn(`AI: reasoning '${opts.reasoning}' is honoured only by lmx engines — `
144
+ + `${cfg.label || cfg.provider} sends no reasoning flag and runs on its own default `
145
+ + '(warned once per provider).');
146
+ }
147
+ const { reasoning: _ignored, ...rest } = opts;
148
+ callOpts = rest;
149
+ }
150
+
151
+ const started = Date.now();
152
+ let result;
153
+ try {
154
+ result = await withOneRetry(() => adapter.complete(cfg, callOpts), opts);
155
+ } catch (err) {
156
+ // A FAILED CALL CAN STILL HAVE SPENT TOKENS (0.26.1) — a model that thought until it ran out of
157
+ // room, a structured reply cut off mid-bracket. Recording only successes let a benchmark or a
158
+ // connection test on such a model run outside the ceiling. Adapters put normalised usage on
159
+ // the error; a failure that spent nothing (refused, unreachable) carries none and is not counted.
160
+ // The ledger failing must never replace the error the caller needs to see.
161
+ if (err && err.usage && err.usage.total > 0) {
162
+ try {
163
+ await meter.record(cfg, { usage: err.usage, model: cfg.model, ms: Date.now() - started, failed: true },
164
+ { ticketId: opts.ticketId });
165
+ } catch (recErr) {
166
+ log.warn(`AI: usage of a failed call not recorded (${recErr.message})`);
167
+ }
168
+ }
169
+ throw err;
170
+ }
171
+
172
+ // ...and the counter after it, AWAITED: two calls in quick succession must both be counted
173
+ // before the second's ceiling check reads the total, or the limit is enforced against a stale one.
174
+ await meter.record(cfg, result, { ticketId: opts.ticketId });
175
+
176
+ log.info(`AI: ${cfg.provider}/${result.model} ${result.usage.total} tokens in ${result.ms}ms`);
177
+ return result;
178
+ }
179
+
180
+ /**
181
+ * Embed one or more strings, with the same retry policy as complete() — a fast dropped connection
182
+ * is retried once; a 429 fails fast (0.25.0, see withOneRetry). Embeddings are idempotent, so the
183
+ * one resend is always safe.
184
+ *
185
+ * THE CONFIG IS THE CALLER'S. An app resolves its embedding settings separately from completion
186
+ * (embeddingModel, often a different endpoint), so this never falls back to resolveConfig().
187
+ * No budget metering - embeddings were never metered, and starting is a separate decision.
188
+ * @param {object} cfg the resolved embedding config (provider, embeddingModel, …)
189
+ * @param {string|string[]} texts
190
+ * @param {{signal?: AbortSignal}} [opts]
191
+ */
192
+ async function embed(cfg, texts, opts = {}) {
193
+ // EXPLICITLY DISABLED is refused, as complete() refuses it (the review found embed ignored it).
194
+ // Strictly `false`: an embedding config the caller built without the flag is still honoured.
195
+ if (cfg && cfg.enabled === false) {
196
+ throw new AiError('disabled', 'This endpoint is switched off.');
197
+ }
198
+ const adapter = cfg && PROVIDERS[cfg.provider];
199
+ if (!adapter) {
200
+ throw new AiError('unconfigured', `No adapter is registered for provider "${cfg && cfg.provider}".`);
201
+ }
202
+ if (typeof adapter.embed !== 'function') {
203
+ throw new AiError('unsupported', `${cfg.label || cfg.provider} does not support embeddings.`);
204
+ }
205
+ return withOneRetry(() => adapter.embed(cfg, texts), opts);
206
+ }
207
+
208
+ /** Is there a provider configured at all? Callers use this to decide whether to render a control. */
209
+ async function isEnabled() {
210
+ return (await resolveConfig()).enabled;
211
+ }
212
+
213
+ /** A short round trip for an admin "Test connection". NEVER THROWS — it reports what is wrong. */
214
+ async function test(cfgOverride) {
215
+ const cfg = cfgOverride || await resolveConfig();
216
+ if (!cfg.enabled) return { ok: false, kind: 'disabled', error: 'No provider is selected.' };
217
+ try {
218
+ const r = await complete({
219
+ system: 'Reply with the single word: ready. Do not explain.',
220
+ messages: [{ role: 'user', content: 'ready?' }],
221
+ maxTokens: 512,
222
+ temperature: 0,
223
+ skipBudget: true // an admin diagnosing a provider must not be blocked by a full budget
224
+ }, cfg);
225
+ return {
226
+ ok: true, model: r.model, ms: r.ms, reply: (r.text || '').trim().slice(0, 60), usage: r.usage,
227
+ finishReason: r.finishReason || null, reasoned: !!r.reasonedFor
228
+ };
229
+ } catch (err) {
230
+ if (err instanceof AiError) return { ok: false, kind: err.kind, error: err.message, retryable: err.retryable };
231
+ return { ok: false, kind: 'bad_response', error: err.message };
232
+ }
233
+ }
234
+
235
+ /** What the server has loaded, for an admin model picker. Empty when it cannot say. */
236
+ async function listModels(cfgOverride) {
237
+ return (await listModelsResult(cfgOverride)).models;
238
+ }
239
+
240
+ /**
241
+ * The model list and WHY it is the length it is.
242
+ *
243
+ * Callers that only want names should use listModels(). An admin screen wants this one: an empty
244
+ * array on its own cannot tell "the server has nothing loaded" from "that address is not an
245
+ * OpenAI-compatible API root", and those need different fixes.
246
+ */
247
+ async function listModelsResult(cfgOverride) {
248
+ const cfg = cfgOverride || await resolveConfig();
249
+ if (!cfg.enabled) {
250
+ return { ok: false, models: [], url: null, status: 0, error: 'No provider is selected.' };
251
+ }
252
+ const adapter = PROVIDERS[cfg.provider];
253
+ if (!adapter) {
254
+ return { ok: false, models: [], url: null, status: 0, error: `Unknown provider “${cfg.provider}”.` };
255
+ }
256
+ try {
257
+ if (adapter.listModelsResult) return await adapter.listModelsResult(cfg);
258
+ // An adapter that predates this contract still works; it simply cannot explain itself.
259
+ return { ok: true, models: await adapter.listModels(cfg), url: null, status: 0, error: null };
260
+ } catch (err) {
261
+ return { ok: false, models: [], url: null, status: 0, error: err.message };
262
+ }
263
+ }
264
+
265
+ /**
266
+ * A fixed-workload speed test for one endpoint. See benchmark.js for why the workload is fixed
267
+ * rather than the prompt, and why the warm-up is reported rather than discarded.
268
+ */
269
+ async function benchmarkEndpoint(cfgOverride, benchOpts) {
270
+ const cfg = cfgOverride || await resolveConfig();
271
+ return require('./benchmark').benchmark(complete, cfg, benchOpts || {});
272
+ }
273
+
274
+ // The writing-assist engines are bound to this client's complete() so a caller gets config +
275
+ // budget + retry for free. Prompt framing is supplied per call by the app (content).
276
+ const boundPolish = (opts) => polish(complete, opts);
277
+ const boundGenerate = (opts) => generate(complete, opts);
278
+
279
+ return {
280
+ complete, embed, isEnabled, test, listModels, listModelsResult, withOneRetry,
281
+ benchmark: benchmarkEndpoint,
282
+ polish: boundPolish, generate: boundGenerate,
283
+ facts, AiError, PROVIDERS, DEFAULTS
284
+ };
285
+ }
286
+
287
+ module.exports = {
288
+ // The usage counter behind every ceiling. LAZY: it needs the db-worker driver contract, which
289
+ // is an OPTIONAL peer — a consumer using only createAiClient/polish/facts must not be made to
290
+ // install a database package to require this one.
291
+ get createUsageStore() { return require('./usageStore').createUsageStore; },
292
+ get createProviderStore() { return require('./providerStore').createProviderStore; },
293
+ // Speed history. Lazy for the same reason as the others: it needs the db-worker driver contract,
294
+ // which is an optional peer.
295
+ get createSpeedStore() { return require('./speedStore').createSpeedStore; },
296
+ // EXPORTED SO A CONSUMER CAN ASSERT ITS TABLE MATCHES. An app writes its own migration, which is
297
+ // a hand copy of this DDL — and a copy with nothing comparing it to the original is the failure
298
+ // mode this repo has already documented twice. providerSchemaFor and usageSchemaFor exist for the
299
+ // same reason; leaving this one out meant a drift would surface as an INSERT throwing at runtime.
300
+ get speedSchemaFor() { return require('./speedStore').schemaFor; },
301
+ // No database behind health, so it loads eagerly like the rest of the seam.
302
+ ...require('./health'),
303
+ /**
304
+ * Where this package's EJS partials live, for the consumer's view-roots list.
305
+ *
306
+ * Same contract as backup/server/notify/uploads: the package knows its own layout, the app
307
+ * puts its own views FIRST so a local file of the same name wins.
308
+ */
309
+ viewsDir: require('path').join(__dirname, 'views'),
310
+ get providerSchemaFor() { return require('./providerStore').schemaFor; },
311
+ // ── SUPERVISED STACKS (lmx) ─────────────────────────────────────────────────────────────────
312
+ // The store is LAZY for the same reason as the others: it needs the db-worker driver contract,
313
+ // which is an optional peer. A consumer using only createAiClient must not be made to install a
314
+ // database package to require this one.
315
+ get createLmxStore() { return require('./lmxStore').createLmxStore; },
316
+ // EXPORTED SO A CONSUMER CAN ASSERT ITS TABLE MATCHES — the migration is a hand copy of this, and
317
+ // a copy with nothing comparing it to the original is the failure this repo has documented three
318
+ // times now. The first app to carry the table wrote it with nothing to check against.
319
+ get lmxSchemaFor() { return require('./lmxStore').schemaFor; },
320
+ // Does this stack actually work? Four checks in the only order they can run. No database and no
321
+ // keystore — every credential arrives as an argument — so it loads eagerly.
322
+ lmxVerify: require('./lmxVerify'),
323
+ // Engine identity - (instance, id), falling back to name - shared by the router, the stack panel
324
+ // and the verifier. Apps used to deep-require providers/lmxDiscovery for it; that path still
325
+ // works, but this is the supported one.
326
+ findEngine: require('./providers/lmxDiscovery').findEngine,
327
+ // The counting rules a screen needs. Separate from the verifier because "what did the stack say"
328
+ // and "what does that mean for what I am relying on" are different questions, and only the second
329
+ // one needs to know which engines this app has adopted.
330
+ ...require('./lmxStatus'),
331
+ get usageSchemaFor() { return require('./usageStore').schemaFor; },
332
+ createAiClient,
333
+ PROVIDERS, DEFAULTS,
334
+ AiError, fromFetchFailure, redact,
335
+ facts,
336
+ // THE INPUT SIDE of the same concern facts.js covers on the output side: text somebody else
337
+ // wrote, placed where a model can read it without being able to give orders. See
338
+ // untrusted.js for why the fence marker has to be generated per call.
339
+ untrusted: require('./untrusted'),
340
+ // ...and the correct ASSEMBLY of it. untrusted.js hands over `fence()` and `rule()` separately;
341
+ // fenced.js puts them together, because one app assembled them two different ways in two prompt
342
+ // builders and one of the divergences started refusing correct answers. See fenced.js for the
343
+ // measurements, including why a fenced document is worth about a third of the defence unless the
344
+ // caller also states its schema's field names.
345
+ fenced: require('./fenced'),
346
+ // Default writing-op catalogues, so an app can build its menus without re-declaring them.
347
+ POLISH_MODES: require('./polish').MODES,
348
+ POLISH_TONES: require('./polish').TONES,
349
+ // The values complete({ reasoning }) accepts (0.26.0), for a caller that validates its own config.
350
+ REASONING_MODES,
351
+ RETRY_AFTER_MS
352
+ };