@aria-framework/ai 0.4.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/health.js CHANGED
@@ -12,9 +12,19 @@
12
12
  * Reporting that as "healthy" hides the only fact worth knowing. So `reachable` and `modelPresent`
13
13
  * are separate, and the UI shows them separately.
14
14
  *
15
- * `modelPresent` is NULL, not false, where residency has no meaning — a hosted API does not load
16
- * models on demand, and flagging one as "not resident" would be a false alarm on a working
17
- * provider. Null means "not applicable", false means "we asked and it is not there".
15
+ * `modelPresent` is NULL, not false, whenever the answer is not KNOWN — a hosted API does not load
16
+ * models on demand, and a server that cannot report load state must not be guessed at. Null means
17
+ * "cannot tell", false means "we asked something that knows, and it said no".
18
+ *
19
+ * ── /v1/models CANNOT ANSWER THIS, WHICH TOOK REAL HARDWARE TO FIND ─────────────────────────────
20
+ * Measured against a live LM Studio: `/v1/models` returned FIVE models with keys `id, object,
21
+ * owned_by`, while only TWO were actually resident. It lists what is DOWNLOADED, not what is
22
+ * LOADED — so residency judged from it reports "loaded" for every model on the disk, and the
23
+ * evicted-model warning this module exists for could never fire at all.
24
+ *
25
+ * LM Studio's native `/api/v0/models` carries a `state` field (`loaded` / `not-loaded`), which is
26
+ * the real signal. So the probe asks that FIRST for kinds that load on demand, and falls back to
27
+ * `/v1/models` purely for reachability — reporting residency as null rather than inventing it.
18
28
  *
19
29
  * ── THE CIRCUIT BREAKER IS WHY THIS IS NOT JUST A PING ──────────────────────────────────────────
20
30
  * Without one, every call to a dead provider pays the full timeout before failing over. Annoying at
@@ -35,8 +45,19 @@ const DEFAULT_TIMEOUT_MS = 8000;
35
45
  /** Kinds that load models on demand, where residency is a real question. */
36
46
  const RESIDENT_KINDS = new Set(['lmstudio', 'openai-compatible']);
37
47
 
38
- /** Strip a trailing slash so `${base}/models` never becomes `//models`. */
39
- const apiRoot = (baseUrl) => String(baseUrl || '').replace(/\/+$/, '');
48
+ /**
49
+ * THE ADAPTER'S root, not a second one.
50
+ *
51
+ * This file originally rolled its own — strip the trailing slash and append `/models` — which is
52
+ * wrong for the base URL people actually type. `http://host:1234` has no path, so the adapter adds
53
+ * `/v1`; the probe did not, and asked for `/models`. LM Studio answers that with
54
+ * "Unexpected endpoint or method. Returning 200 anyway" — a 200 with no model list — so the probe
55
+ * saw an empty listing and reported a perfectly loaded model as EVICTED, on a server that was
56
+ * answering completions the whole time.
57
+ *
58
+ * Two roots that disagree about where the API lives is a bug waiting to happen twice. There is one.
59
+ */
60
+ const { apiRoot } = require('./providers/openai-compatible');
40
61
 
41
62
  /**
42
63
  * The default probe: ask the endpoint what models it has.
@@ -45,6 +66,39 @@ const apiRoot = (baseUrl) => String(baseUrl || '').replace(/\/+$/, '');
45
66
  * for a dead host and [] for a live host with nothing loaded, which is exactly the difference this
46
67
  * module exists to report.
47
68
  */
69
+ /**
70
+ * LM Studio's native model list, which reports LOAD STATE.
71
+ *
72
+ * Returns null when the endpoint is not there (any other server), so the caller falls back rather
73
+ * than treating its absence as a failure.
74
+ */
75
+ async function nativeResidency(cfg, fetchImpl, signalMs) {
76
+ // The native API sits beside the OpenAI-compatible one, not under /v1.
77
+ let base;
78
+ try {
79
+ const u = new URL(String(cfg.baseUrl || ''));
80
+ base = `${u.protocol}//${u.host}`;
81
+ } catch (_) { return null; }
82
+
83
+ try {
84
+ const res = await fetchImpl(`${base}/api/v0/models`, {
85
+ headers: cfg.apiKey ? { Authorization: `Bearer ${cfg.apiKey}` } : {},
86
+ signal: AbortSignal.timeout(signalMs)
87
+ });
88
+ if (!res.ok) return null;
89
+ const payload = await res.json();
90
+ if (!Array.isArray(payload && payload.data)) return null;
91
+ const loaded = payload.data
92
+ .filter((m) => m && m.state === 'loaded')
93
+ .map((m) => m.id || m.key || '')
94
+ .filter(Boolean);
95
+ // An empty `data` is a real answer (nothing loaded); a missing one was handled above.
96
+ return { loaded, all: payload.data.map((m) => (m && m.id) || '').filter(Boolean) };
97
+ } catch (_) {
98
+ return null;
99
+ }
100
+ }
101
+
48
102
  async function defaultProbe(cfg, { fetchImpl = fetch, timeoutMs } = {}) {
49
103
  const url = `${apiRoot(cfg.baseUrl)}/models`;
50
104
  const started = Date.now();
@@ -62,25 +116,47 @@ async function defaultProbe(cfg, { fetchImpl = fetch, timeoutMs } = {}) {
62
116
  // lives in `e.cause.code`, so lead with that and keep the message as context.
63
117
  const cause = (e && e.cause) || {};
64
118
  const detail = cause.code || (e && e.message) || 'unreachable';
119
+ // NAME THE URL. Nothing normalises a base URL any more, so a wrong one is a real possibility
120
+ // and the operator's first question is "what did it actually call". `ECONNREFUSED` alone sends
121
+ // them to check the server; `ECONNREFUSED at http://host:1234/models` shows them the typo.
122
+ const reason = timedOut
123
+ ? `no response within ${timeoutMs || cfg.timeoutMs || DEFAULT_TIMEOUT_MS}ms`
124
+ : (cause.code ? `${cause.code}${cause.message ? ' — ' + cause.message : ''}` : detail);
65
125
  return {
66
- reachable: false, models: [], ms: Date.now() - started,
67
- error: timedOut
68
- ? `no response within ${timeoutMs || cfg.timeoutMs || DEFAULT_TIMEOUT_MS}ms`
69
- : (cause.code ? `${cause.code}${cause.message ? ' — ' + cause.message : ''}` : detail)
126
+ reachable: false, listed: false, models: [], loaded: null, ms: Date.now() - started,
127
+ url, error: `${reason} (${url})`
70
128
  };
71
129
  }
72
130
  const ms = Date.now() - started;
73
131
  if (!res.ok) {
74
132
  // It ANSWERED, so the host is reachable — the failure is auth, quota or a bad path, and saying
75
133
  // "unreachable" would send an operator to check the network instead of the key.
76
- return { reachable: true, models: [], ms, error: `HTTP ${res.status}` };
134
+ return { reachable: true, listed: false, models: [], loaded: null, ms, url, error: `HTTP ${res.status} (${url})` };
77
135
  }
78
136
  let payload = null;
79
137
  try { payload = await res.json(); } catch (_) { payload = null; }
80
- const models = Array.isArray(payload && payload.data)
138
+
139
+ // "A 200 THAT IS NOT A MODEL LISTING" IS NOT "NO MODELS LOADED".
140
+ //
141
+ // Servers answer 200 to things they do not implement — LM Studio logs
142
+ // "Unexpected endpoint or method. Returning 200 anyway" and returns a body with no `data`.
143
+ // Treating that as an empty list makes a loaded model look evicted, which is the exact false
144
+ // alarm this module exists to avoid. So `listed` says whether a real listing came back, and
145
+ // residency is only judged when it did.
146
+ const listed = Array.isArray(payload && payload.data);
147
+ const models = listed
81
148
  ? payload.data.map((m) => (m && (m.id || m.name)) || '').filter(Boolean)
82
149
  : [];
83
- return { reachable: true, models, ms, error: null };
150
+
151
+ // Now ask something that actually knows about residency. Only for kinds that load on demand,
152
+ // and only as an addition — a server without it still reports reachable, with residency null.
153
+ let loaded = null;
154
+ if (RESIDENT_KINDS.has(cfg.provider)) {
155
+ const native = await nativeResidency(cfg, fetchImpl, timeoutMs || cfg.timeoutMs || DEFAULT_TIMEOUT_MS);
156
+ if (native) loaded = native.loaded;
157
+ }
158
+
159
+ return { reachable: true, listed, models, loaded, ms, url, error: null };
84
160
  }
85
161
 
86
162
  /**
@@ -166,9 +242,13 @@ function createHealthChecker(opts = {}) {
166
242
 
167
243
  const r = await probe(cfg);
168
244
  // Residency only means something where models are loaded on demand.
245
+ // RESIDENCY COMES FROM `loaded`, NEVER FROM THE MODEL LIST. `/v1/models` reports what is
246
+ // downloaded; judging residency from it says "loaded" for every model on the disk. Null
247
+ // whenever nothing authoritative answered — an unknown is honest, a guess is not.
169
248
  let modelPresent = null;
170
- if (RESIDENT_KINDS.has(cfg.provider) && r.reachable && !r.error) {
171
- modelPresent = cfg.model ? r.models.includes(cfg.model) : null;
249
+ if (RESIDENT_KINDS.has(cfg.provider) && r.reachable && !r.error
250
+ && Array.isArray(r.loaded) && cfg.model) {
251
+ modelPresent = r.loaded.includes(cfg.model);
172
252
  }
173
253
 
174
254
  // REACHABLE BUT NOT LOADED IS NOT A FAILURE. The call will still succeed; it will just be
package/index.js CHANGED
@@ -177,6 +177,13 @@ module.exports = {
177
177
  get createProviderStore() { return require('./providerStore').createProviderStore; },
178
178
  // No database behind health, so it loads eagerly like the rest of the seam.
179
179
  ...require('./health'),
180
+ /**
181
+ * Where this package's EJS partials live, for the consumer's view-roots list.
182
+ *
183
+ * Same contract as backup/server/notify/uploads: the package knows its own layout, the app
184
+ * puts its own views FIRST so a local file of the same name wins.
185
+ */
186
+ viewsDir: require('path').join(__dirname, 'views'),
180
187
  get providerSchemaFor() { return require('./providerStore').schemaFor; },
181
188
  get usageSchemaFor() { return require('./usageStore').schemaFor; },
182
189
  createAiClient,
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@aria-framework/ai",
3
3
  "description": "Aria App Framework — AI module. A dependency-injected model seam (createAiClient) over several providers (LM Studio / OpenAI-compatible / Anthropic), with a fact-preservation guard, generic Polish and Generate writing engines, and a browser polish widget. Prompts and config stay in the consuming app.",
4
- "version": "0.4.0",
4
+ "version": "0.7.0",
5
5
  "license": "UNLICENSED",
6
6
  "private": false,
7
7
  "publishConfig": {
@@ -16,7 +16,7 @@
16
16
  "generate.js",
17
17
  "providers/openai-compatible.js",
18
18
  "providers/anthropic.js",
19
- "browser/ai-polish.js", "usageStore.js", "providerStore.js", "health.js"
19
+ "browser/ai-polish.js", "usageStore.js", "providerStore.js", "health.js", "views/"
20
20
  ],
21
21
  "peerDependencies": {
22
22
  "@aria-framework/db-worker": ">=0.7.0"
@@ -25,6 +25,6 @@
25
25
  "@aria-framework/db-worker": { "optional": true }
26
26
  },
27
27
  "scripts": {
28
- "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/health.js"
28
+ "test": "node test/smoke.js && node test/usageStore.js && node test/providerStore.js && node test/health.js && node test/views.js"
29
29
  }
30
30
  }
@@ -251,15 +251,29 @@ async function httpError(res, label, apiKey) {
251
251
  * left alone, because other servers in this family mount elsewhere (LiteLLM behind a prefix, for
252
252
  * one) and second-guessing an explicit path would break them.
253
253
  */
254
+ /**
255
+ * The configured base URL, with a trailing slash trimmed. Nothing else.
256
+ *
257
+ * IT USED TO APPEND `/v1` WHEN THE URL HAD NO PATH, and that was wrong for reasons that only got
258
+ * clearer the more providers there were:
259
+ *
260
+ * - the stored value and the called value differed, so the form did not show what went on the
261
+ * wire;
262
+ * - it cannot generalise. One provider wants `/v1`, another `/api/v1`, another nothing at all.
263
+ * A blanket rule is wrong for every provider it was not written for, and there is no way to
264
+ * know which is which;
265
+ * - it turned a typo into a SILENT wrong answer that surfaced three layers away. A base URL
266
+ * missing `/v1` still completed chats — because this quietly fixed it — while the health probe
267
+ * asked a path LM Studio answers with "Unexpected endpoint, returning 200 anyway", and a
268
+ * perfectly loaded model was reported as evicted.
269
+ *
270
+ * So: call exactly what was configured. A wrong URL now fails immediately and says which URL it
271
+ * called, which is a configuration error an operator can act on rather than a mystery.
272
+ *
273
+ * Trailing-slash trimming stays because it changes no meaning — it only prevents `//models`.
274
+ */
254
275
  function apiRoot(baseUrl) {
255
- const trimmed = String(baseUrl || '').replace(/\/+$/, '');
256
- let path;
257
- try {
258
- path = new URL(trimmed).pathname.replace(/\/+$/, '');
259
- } catch (err) {
260
- return trimmed; // not a URL we can parse; leave it exactly as typed
261
- }
262
- return path === '' ? trimmed + '/v1' : trimmed;
276
+ return String(baseUrl || '').replace(/\/+$/, '');
263
277
  }
264
278
 
265
279
  /** Both providers report tokens; they name the fields differently. One shape reaches the caller. */
@@ -0,0 +1,138 @@
1
+ <%#
2
+ One provider: what it is, whether it works, and what it has cost.
3
+
4
+ LOCALS
5
+ p a row from createProviderStore.all()
6
+ health a status object from createHealthChecker.status(p.id) — may be undefined
7
+ usage { calls, total_tokens } for the window, or undefined
8
+ canEdit boolean — hides controls rather than disabling them, because a greyed-out
9
+ button still advertises a feature and invites a support ticket
10
+ routes [{ id, position }] this provider appears in, for the shared-fallback warning
11
+
12
+ WHY "reachable" AND "model loaded" ARE SHOWN SEPARATELY: a local server answers /models
13
+ perfectly while the model itself has been evicted, and the next call then pays a
14
+ multi-second reload. One green dot would hide the only fact worth knowing.
15
+ -%>
16
+ <%
17
+ const h = typeof health !== 'undefined' && health ? health : { status: 'unknown', failures: 0 };
18
+ const local = p.kind === 'lmstudio' || p.kind === 'openai-compatible';
19
+ const dot = !p.enabled ? 'secondary'
20
+ : h.status === 'down' ? 'danger'
21
+ : h.status === 'up' ? (h.modelPresent === false ? 'warning' : 'success')
22
+ : 'secondary';
23
+ const ago = (ts) => {
24
+ if (!ts) return null;
25
+ const s = Math.max(0, Math.round((Date.now() - ts) / 1000));
26
+ if (s < 60) return s + 's ago';
27
+ if (s < 3600) return Math.round(s / 60) + 'm ago';
28
+ if (s < 86400) return Math.round(s / 3600) + 'h ago';
29
+ return Math.round(s / 86400) + 'd ago';
30
+ };
31
+ const shared = (typeof routes !== 'undefined' && routes) ? routes : [];
32
+ -%>
33
+ <article class="card mb-3">
34
+ <div class="card-body d-flex flex-wrap gap-3 justify-content-between align-items-start">
35
+ <div class="flex-grow-1" style="min-width:18rem">
36
+
37
+ <div class="d-flex align-items-center gap-2 mb-1">
38
+ <span class="badge rounded-pill bg-<%= dot %>" style="width:.6rem;height:.6rem;padding:0"
39
+ aria-hidden="true"></span>
40
+ <strong class="fs-6"><%= p.id %></strong>
41
+ <span class="badge text-bg-light border"><%= p.kind %></span>
42
+ <% if (local) { %><span class="badge text-bg-light border">local</span><% } %>
43
+ <% if (!p.enabled) { %><span class="badge text-bg-secondary">disabled</span><% } %>
44
+ <% if (h.status === 'down') { %>
45
+ <span class="badge text-bg-danger">
46
+ down<%= h.cooldownRemainingMs ? ' · retrying in ' + Math.ceil(h.cooldownRemainingMs / 1000) + 's' : '' %>
47
+ </span>
48
+ <% } else if (h.modelPresent === false) { %>
49
+ <span class="badge text-bg-warning">model not loaded</span>
50
+ <% } %>
51
+ </div>
52
+
53
+ <div class="text-body-secondary small font-monospace mb-2">
54
+ <%= p.model || '(no chat model)' %><% if (p.embedding_model) { %> · <%= p.embedding_model %><% } %>
55
+ <% if (p.base_url) { %> · <%= p.base_url %><% } %>
56
+ </div>
57
+
58
+ <div class="d-flex flex-wrap gap-4 small">
59
+ <% if (p.context_tokens) { %>
60
+ <div><div class="text-body-secondary text-uppercase" style="font-size:.68rem;letter-spacing:.06em">Context</div>
61
+ <span class="font-monospace"><%= Number(p.context_tokens).toLocaleString() %></span></div>
62
+ <% } %>
63
+ <% if (h.lastMs != null) { %>
64
+ <div><div class="text-body-secondary text-uppercase" style="font-size:.68rem;letter-spacing:.06em">Latency</div>
65
+ <span class="font-monospace"><%= h.lastMs %> ms</span></div>
66
+ <% } %>
67
+ <% if (typeof usage !== 'undefined' && usage) { %>
68
+ <div><div class="text-body-secondary text-uppercase" style="font-size:.68rem;letter-spacing:.06em">Calls</div>
69
+ <span class="font-monospace"><%= Number(usage.calls || 0).toLocaleString() %></span></div>
70
+ <div><div class="text-body-secondary text-uppercase" style="font-size:.68rem;letter-spacing:.06em">Tokens</div>
71
+ <span class="font-monospace"><%= Number(usage.total_tokens || 0).toLocaleString() %></span></div>
72
+ <% } %>
73
+ <div>
74
+ <%# LAST GOOD, not just a dot. "last good 3h ago" tells a story a green light cannot. %>
75
+ <div class="text-body-secondary text-uppercase" style="font-size:.68rem;letter-spacing:.06em">Last good</div>
76
+ <span class="font-monospace"><%= ago(h.lastGoodAt) || 'never' %></span>
77
+ </div>
78
+ </div>
79
+
80
+ <% if (p.daily_token_cap) { %>
81
+ <%
82
+ const spent = (typeof usage !== 'undefined' && usage) ? Number(usage.total_tokens || 0) : 0;
83
+ const pct = Math.min(100, Math.round((spent / Number(p.daily_token_cap)) * 100));
84
+ -%>
85
+ <div class="mt-3" style="max-width:26rem">
86
+ <div class="d-flex justify-content-between small text-body-secondary">
87
+ <span>Daily token cap</span>
88
+ <span class="font-monospace"><%= spent.toLocaleString() %> / <%= Number(p.daily_token_cap).toLocaleString() %></span>
89
+ </div>
90
+ <div class="progress" style="height:.4rem" role="progressbar" aria-valuenow="<%= pct %>"
91
+ aria-valuemin="0" aria-valuemax="100">
92
+ <div class="progress-bar bg-<%= pct >= 100 ? 'danger' : pct >= 80 ? 'warning' : 'success' %>"
93
+ style="width:<%= pct %>%"></div>
94
+ </div>
95
+ </div>
96
+ <% } %>
97
+
98
+ <% if (h.status === 'down' && h.lastError) { %>
99
+ <div class="alert alert-danger d-flex gap-2 mt-3 mb-0 py-2">
100
+ <i class="bi bi-exclamation-triangle mt-1"></i>
101
+ <div class="small">
102
+ <strong>Marked down after <%= h.failures %> consecutive failure<%= h.failures === 1 ? '' : 's' %>.</strong>
103
+ Last error: <code><%= h.lastError %></code>. Calls skip this provider entirely rather than
104
+ paying its timeout, and it returns to service on the first success.
105
+ </div>
106
+ </div>
107
+ <% } else if (h.modelPresent === false) { %>
108
+ <div class="alert alert-warning d-flex gap-2 mt-3 mb-0 py-2">
109
+ <i class="bi bi-hourglass-split mt-1"></i>
110
+ <div class="small">
111
+ <strong>Reachable, but <code><%= p.model %></code> is not loaded.</strong>
112
+ The next call pays a reload before it does anything. This is not a failure and nothing
113
+ fails over — but if it keeps happening, turn off automatic model unloading on the server.
114
+ </div>
115
+ </div>
116
+ <% } %>
117
+
118
+ <% if (shared.length > 1) { %>
119
+ <div class="alert alert-warning d-flex gap-2 mt-3 mb-0 py-2">
120
+ <i class="bi bi-diagram-3 mt-1"></i>
121
+ <div class="small">
122
+ <strong>Serves <%= shared.length %> routes.</strong>
123
+ <%= shared.map((r) => r.id + ' (' + (r.position === 0 ? 'primary' : 'fallback') + ')').join(', ') %>.
124
+ If it goes down, one route fails over <em>and</em> another loses its safety net at the
125
+ same moment.
126
+ </div>
127
+ </div>
128
+ <% } %>
129
+ </div>
130
+
131
+ <div class="d-flex flex-column gap-2 align-items-end">
132
+ <% if (typeof canEdit !== 'undefined' && canEdit) { %>
133
+ <button class="btn btn-sm btn-outline-secondary" name="test_provider" value="<%= p.id %>">Test</button>
134
+ <a class="btn btn-sm btn-outline-secondary" href="?edit=<%= encodeURIComponent(p.id) %>">Edit</a>
135
+ <% } %>
136
+ </div>
137
+ </div>
138
+ </article>
@@ -0,0 +1,58 @@
1
+ <%#
2
+ Per-provider usage over a window.
3
+
4
+ LOCALS
5
+ rows from createUsageStore.byProvider(sinceDay) — [{provider, model, calls, ...}]
6
+ prices optional { [model]: costPerMillionTokens } supplied BY THE APP. Prices change, and
7
+ a stale hardcoded price silently produces confidently wrong cost reports, so the
8
+ package never carries one.
9
+ kinds optional { [providerId]: kind } so a local endpoint can be shown as having no cost
10
+ rather than a cost of zero — different facts.
11
+ -%>
12
+ <%
13
+ const priceFor = (m) => (typeof prices !== 'undefined' && prices && prices[m] != null) ? Number(prices[m]) : null;
14
+ const kindOf = (id) => (typeof kinds !== 'undefined' && kinds) ? kinds[id] : undefined;
15
+ const isLocal = (id) => ['lmstudio', 'openai-compatible'].includes(kindOf(id));
16
+ const n = (v) => Number(v || 0).toLocaleString();
17
+ -%>
18
+ <div class="table-responsive">
19
+ <table class="table table-sm align-middle mb-0">
20
+ <thead>
21
+ <tr class="small text-body-secondary text-uppercase">
22
+ <th>Provider</th><th>Model</th>
23
+ <th class="text-end">Calls</th><th class="text-end">Prompt</th>
24
+ <th class="text-end">Completion</th><th class="text-end">Total</th><th class="text-end">Cost</th>
25
+ </tr>
26
+ </thead>
27
+ <tbody>
28
+ <% if (!rows || !rows.length) { %>
29
+ <tr><td colspan="7" class="text-body-secondary py-3">Nothing recorded in this window.</td></tr>
30
+ <% } %>
31
+ <% (rows || []).forEach(function (r) {
32
+ const price = priceFor(r.model);
33
+ const cost = price == null ? null : (Number(r.total_tokens || 0) / 1e6) * price;
34
+ -%>
35
+ <tr>
36
+ <td><%= r.provider %></td>
37
+ <td class="font-monospace small"><%= r.model %></td>
38
+ <td class="text-end font-monospace"><%= n(r.calls) %></td>
39
+ <td class="text-end font-monospace"><%= n(r.prompt_tokens) %></td>
40
+ <td class="text-end font-monospace"><%= n(r.completion_tokens) %></td>
41
+ <td class="text-end font-monospace"><%= n(r.total_tokens) %></td>
42
+ <td class="text-end font-monospace">
43
+ <%# A local GPU has NO cost, which is a different fact from a cost of zero. Showing
44
+ "0.00" would invite someone to compare it against a cloud row as if it were cheap. %>
45
+ <% if (isLocal(r.provider)) { %><span class="text-body-secondary">&mdash;</span>
46
+ <% } else if (cost == null) { %><span class="text-body-secondary" title="No price configured for this model">?</span>
47
+ <% } else { %><%= cost.toFixed(2) %><% } %>
48
+ </td>
49
+ </tr>
50
+ <% }); -%>
51
+ </tbody>
52
+ </table>
53
+ </div>
54
+ <p class="small text-body-secondary mt-2 mb-0">
55
+ Tokens are recorded for every call, including local models. Cost is derived from tokens and the
56
+ price table this app supplies &mdash; never stored &mdash; so a local endpoint still reports real
57
+ numbers, just different ones: throughput and utilisation rather than spend.
58
+ </p>