@dotdrelle/wiki-manager 0.15.34 → 0.15.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/.env.example +8 -5
  2. package/README.md +42 -0
  3. package/agents.docker-compose.yml +11 -16
  4. package/docker-compose.yml +9 -8
  5. package/package.json +2 -2
  6. package/src/agent/graph.js +20 -0
  7. package/src/cli/wiki-manager.js +20 -0
  8. package/src/commands/slash.js +85 -13
  9. package/src/commands/slash.test.js +85 -3
  10. package/src/core/agentsCompose.js +86 -1
  11. package/src/core/buildInfo.json +2 -2
  12. package/src/core/compose.js +24 -2
  13. package/src/core/dockerCompose.test.js +18 -0
  14. package/src/core/env.js +2 -1
  15. package/src/core/env.test.js +14 -0
  16. package/src/core/googleGrants.js +38 -0
  17. package/src/core/googleGrants.test.js +59 -0
  18. package/src/core/mcp.js +1 -1
  19. package/src/core/mcp.test.js +1 -0
  20. package/src/core/mcpEndpoints.js +96 -0
  21. package/src/core/mcpEndpoints.test.js +126 -0
  22. package/src/core/modelFetch.js +241 -27
  23. package/src/core/modelFetch.test.js +213 -0
  24. package/src/core/profileServiceStatus.test.js +146 -0
  25. package/src/core/wikiSetup.js +11 -2
  26. package/src/core/wikiWorkspace.test.js +16 -3
  27. package/src/orchestrator/agentRegistry.js +13 -1
  28. package/src/orchestrator/agentRegistry.test.js +20 -0
  29. package/src/runtime/lifecycle.js +105 -3
  30. package/src/runtime/lifecycle.test.js +68 -0
  31. package/src/runtime/server.js +28 -0
  32. package/src/shell/LeftPane.tsx +55 -41
  33. package/src/shell/SetupWizard.tsx +416 -116
  34. package/src/shell/repl.js +61 -12
  35. package/src/shell/repl.test.js +48 -19
  36. package/src/shell/setupWizardDiscovery.test.js +48 -0
  37. package/src/shell/setupWizardPlaceholders.test.js +15 -2
  38. package/src/shell/setupWizardSuggestions.test.js +36 -2
  39. package/src/shell/wrapText.js +57 -0
  40. package/src/shell/wrapText.test.js +48 -0
  41. package/wiki-workspace +122 -32
@@ -149,18 +149,159 @@ function headersFor(provider, engine, apiKey) {
149
149
  return { Authorization: `Bearer ${apiKey}` };
150
150
  }
151
151
 
152
+ /**
153
+ * Type d'un modèle, quand la réponse le porte.
154
+ *
155
+ * `/v1/models` d'OpenAI ne type rien (`object: "model"` partout) — d'où le
156
+ * repli non typé. Mais plusieurs serveurs OpenAI-compatibles ajoutent un
157
+ * champ : Albert annonce `text-generation`, `text-embeddings-inference` ou
158
+ * `text-classification` (son reranker), LiteLLM porte `model_info.mode`. Les
159
+ * ignorer forçait le wizard à proposer les modèles de chat pour l'étape
160
+ * embeddings — sur Albert, aucune suggestion ne pouvait correspondre.
161
+ *
162
+ * Un indice non reconnu (`automatic-speech-recognition`) rend `null` : le
163
+ * modèle est simplement absent des trois listes.
164
+ */
165
+ export function classifyModelEntry(item) {
166
+ const hints = [
167
+ item?.model_info?.mode,
168
+ item?.mode,
169
+ item?.type,
170
+ item?.task,
171
+ item?.object,
172
+ ...(Array.isArray(item?.capabilities) ? item.capabilities : []),
173
+ ]
174
+ .filter((value) => typeof value === 'string')
175
+ .map((value) => value.toLowerCase());
176
+
177
+ for (const hint of hints) {
178
+ if (hint.includes('embed')) return 'embedding';
179
+ // Albert expose son reranker en `text-classification`.
180
+ if (hint.includes('rerank') || hint.includes('classification')) return 'rerank';
181
+ if (hint.includes('chat') || hint.includes('generation') || hint.includes('completion')) {
182
+ return 'chat';
183
+ }
184
+ }
185
+ return null;
186
+ }
187
+
188
+ function itemsOf(provider, engine, payload) {
189
+ return normalizeProvider(provider) === 'openai-compatible' && normalizeEngine(engine) === 'ollama'
190
+ ? payload?.models
191
+ : payload?.data;
192
+ }
193
+
194
+ function sortedUnique(values) {
195
+ return [...new Set(values)].sort((a, b) => a.localeCompare(b));
196
+ }
197
+
152
198
  function parseModelNames(provider, engine, payload) {
153
- const items =
154
- normalizeProvider(provider) === 'openai-compatible' &&
155
- normalizeEngine(engine) === 'ollama'
156
- ? payload?.models
157
- : payload?.data;
199
+ const items = itemsOf(provider, engine, payload);
158
200
  if (!Array.isArray(items)) return [];
159
- return items
160
- .map((item) => item?.id ?? item?.name ?? item?.model)
161
- .filter(Boolean)
162
- .map(String)
163
- .sort((a, b) => a.localeCompare(b));
201
+ return sortedUnique(
202
+ items.map((item) => item?.id ?? item?.name ?? item?.model).filter(Boolean).map(String),
203
+ );
204
+ }
205
+
206
+ /**
207
+ * Délai de découverte du wizard.
208
+ *
209
+ * Un `/v1/models` qui répond le fait en quelques dizaines de millisecondes :
210
+ * ce délai n'est jamais payé par un endpoint sain, il ne borne que les pannes
211
+ * silencieuses (proxy qui avale la connexion, port filtré). Il peut donc
212
+ * rester confortable — la découverte est lancée en tâche de fond, aucune
213
+ * étape du wizard ne l'attend.
214
+ */
215
+ export const DISCOVERY_TIMEOUT_MS = 8000;
216
+
217
+ /** Codes TLS qui désignent une CA privée ou un proxy qui intercepte. */
218
+ const TLS_ERROR_CODES = new Set([
219
+ 'UNABLE_TO_VERIFY_LEAF_SIGNATURE',
220
+ 'SELF_SIGNED_CERT_IN_CHAIN',
221
+ 'DEPTH_ZERO_SELF_SIGNED_CERT',
222
+ 'CERT_HAS_EXPIRED',
223
+ 'CERT_UNTRUSTED',
224
+ 'ERR_TLS_CERT_ALTNAME_INVALID',
225
+ 'UNABLE_TO_GET_ISSUER_CERT_LOCALLY',
226
+ ]);
227
+
228
+ function hostOf(url) {
229
+ try {
230
+ return new URL(url).host;
231
+ } catch {
232
+ return url;
233
+ }
234
+ }
235
+
236
+ function httpStatusHint(status, url) {
237
+ if (status === 401) return `HTTP 401 — API key rejected by ${hostOf(url)}`;
238
+ if (status === 403) {
239
+ return `HTTP 403 — key accepted but access to the model catalog is denied`;
240
+ }
241
+ if (status === 404) return `HTTP 404 — no model catalog exposed at ${url}`;
242
+ if (status === 407) {
243
+ return `HTTP 407 — the HTTP proxy requires authentication (check HTTPS_PROXY credentials)`;
244
+ }
245
+ if (status >= 500) return `HTTP ${status} — the server failed to answer ${url}`;
246
+ return `HTTP ${status} on ${url}`;
247
+ }
248
+
249
+ /**
250
+ * Message actionnable pour un échec réseau.
251
+ *
252
+ * Le message brut de `fetch` ("fetch failed") ne dit rien : la cause utile est
253
+ * dans `err.cause.code`. On la traduit en une phrase qui nomme la manœuvre —
254
+ * proxy, CA privée, port fermé, DNS — parce que c'est exactement ce que
255
+ * l'opérateur doit corriger, et qu'il ne le devinera pas depuis le wizard.
256
+ */
257
+ export function describeFetchError(err, { url, timeoutMs } = {}) {
258
+ if (!err) return 'unknown error';
259
+ if (err.name === 'AbortError' || err.name === 'TimeoutError') {
260
+ // Le délai en millisecondes est un détail d'implémentation : ce qui aide
261
+ // l'opérateur, c'est l'hôte qui n'a pas répondu et les causes probables.
262
+ return `${hostOf(url)} did not answer in time — server unreachable, or blocked by a proxy or firewall`;
263
+ }
264
+ const code = err?.cause?.code ?? err?.code ?? null;
265
+ if (code && TLS_ERROR_CODES.has(code)) {
266
+ return `TLS certificate rejected (${code}) — private CA or intercepting proxy; relaunch with wiki-manager --cacert <file.pem>`;
267
+ }
268
+ if (code === 'ECONNREFUSED') {
269
+ return `connection refused (ECONNREFUSED) by ${hostOf(url)} — nothing is listening on this host/port`;
270
+ }
271
+ if (code === 'ENOTFOUND' || code === 'EAI_AGAIN') {
272
+ return `host not found (${code}): ${hostOf(url)} — check the URL, DNS, or that the proxy resolves it`;
273
+ }
274
+ if (code === 'ETIMEDOUT') {
275
+ return `connection timed out (ETIMEDOUT) to ${hostOf(url)} — usually a firewall dropping the packets`;
276
+ }
277
+ if (code === 'ECONNRESET' || code === 'EPROTO') {
278
+ return `connection reset (${code}) by ${hostOf(url)} — often a proxy intercepting TLS, or http:// used on an https:// endpoint`;
279
+ }
280
+ const message = err instanceof Error ? err.message : String(err);
281
+ return code ? `${message} (${code})` : message;
282
+ }
283
+
284
+ /**
285
+ * État du transport local, affiché à côté d'une erreur de découverte.
286
+ *
287
+ * Un proxy déclaré mais non activé (`NODE_USE_ENV_PROXY` absent) est la panne
288
+ * la plus fréquente en entreprise, et elle est invisible sans ce rappel.
289
+ */
290
+ export function transportSummary(env = process.env) {
291
+ const proxy = env.HTTPS_PROXY ?? env.HTTP_PROXY ?? null;
292
+ const parts = [];
293
+ if (proxy) {
294
+ parts.push(
295
+ env.NODE_USE_ENV_PROXY === '1'
296
+ ? `proxy ${proxy}`
297
+ : `proxy ${proxy} (NODE_USE_ENV_PROXY not set: it is NOT used)`,
298
+ );
299
+ } else {
300
+ parts.push('direct connection (no HTTP(S)_PROXY)');
301
+ }
302
+ const cacert = env.WIKI_MANAGER_CACERT_PATH ?? env.NODE_EXTRA_CA_CERTS ?? null;
303
+ parts.push(cacert ? `CA ${cacert}` : 'CA system trust store');
304
+ return parts.join(' · ');
164
305
  }
165
306
 
166
307
  async function getJson(url, headers, timeoutMs) {
@@ -168,8 +309,11 @@ async function getJson(url, headers, timeoutMs) {
168
309
  const timer = setTimeout(() => controller.abort(), timeoutMs);
169
310
  try {
170
311
  const response = await fetch(url, { headers, signal: controller.signal });
171
- if (!response.ok) throw new Error(`HTTP ${response.status}`);
312
+ if (!response.ok) throw new Error(httpStatusHint(response.status, url));
172
313
  return await response.json();
314
+ } catch (err) {
315
+ if (err instanceof Error && /^HTTP \d/.test(err.message)) throw err;
316
+ throw new Error(describeFetchError(err, { url, timeoutMs }));
173
317
  } finally {
174
318
  clearTimeout(timer);
175
319
  }
@@ -194,7 +338,7 @@ export async function fetchModels(provider, baseUrl, apiKey, options = {}) {
194
338
  };
195
339
  }
196
340
 
197
- const timeoutMs = options.timeoutMs ?? 10000;
341
+ const timeoutMs = options.timeoutMs ?? DISCOVERY_TIMEOUT_MS;
198
342
  try {
199
343
  const needsKey = !(routing === 'openai-compatible' && normalizedEngine === 'ollama');
200
344
  if (needsKey && !apiKey) {
@@ -207,7 +351,12 @@ export async function fetchModels(provider, baseUrl, apiKey, options = {}) {
207
351
  );
208
352
  const models = parseModelNames(provider, normalizedEngine, payload);
209
353
  if (models.length === 0) throw new Error('No models returned');
210
- return { ok: true, models, source: 'remote' };
354
+ // `raw` rend les entrées brutes, seules porteuses des indices de type que
355
+ // `fetchServerCatalog` exploite. Absentes par défaut : la forme historique
356
+ // de ce retour est {ok, models, source}.
357
+ return options.raw
358
+ ? { ok: true, models, source: 'remote', items: itemsOf(provider, normalizedEngine, payload) ?? [] }
359
+ : { ok: true, models, source: 'remote' };
211
360
  } catch (err) {
212
361
  return {
213
362
  ok: false,
@@ -231,10 +380,45 @@ export async function fetchModels(provider, baseUrl, apiKey, options = {}) {
231
380
  * dire à l'utilisateur.
232
381
  * 3. Injoignable : listes vides, `error` renseignée. Le wizard garde sa
233
382
  * saisie libre, qui fait foi de toute façon.
383
+ *
384
+ * Les deux appels partent **en parallèle**, et le résultat est livré en deux
385
+ * temps : `options.onPartial` reçoit la liste plate de `/v1/models` dès
386
+ * qu'elle arrive — c'est la requête rapide, et elle suffit à choisir un
387
+ * modèle — pendant que `/model/info`, plus lourd côté gateway, continue.
388
+ * La promesse résout ensuite avec le catalogue typé s'il aboutit. L'opérateur
389
+ * a donc une liste utilisable immédiatement, qui se raffine sous ses yeux.
234
390
  */
235
391
  export async function fetchGatewayCatalog(baseUrl, apiKey, options = {}) {
236
- const timeoutMs = options.timeoutMs ?? 10000;
392
+ const timeoutMs = options.timeoutMs ?? DISCOVERY_TIMEOUT_MS;
237
393
  const headers = { Authorization: `Bearer ${apiKey}` };
394
+ const onPartial = typeof options.onPartial === 'function' ? options.onPartial : null;
395
+
396
+ const flatPromise = fetchModels('ai-gateway', baseUrl, apiKey, { timeoutMs });
397
+ // Sans ce no-op, un rejet arrivant avant son `await` remonterait en
398
+ // unhandledRejection quand le chemin typé réussit.
399
+ flatPromise.catch(() => {});
400
+
401
+ const flatResult = (flat, error) => ({
402
+ ok: true,
403
+ typed: false,
404
+ source: 'models',
405
+ chat: flat.models,
406
+ embedding: flat.models,
407
+ rerank: flat.models,
408
+ // Conservée pour l'affichage : elle explique pourquoi les listes ne sont
409
+ // pas filtrées.
410
+ ...(error ? { error: error instanceof Error ? error.message : String(error) } : {}),
411
+ });
412
+
413
+ let settled = false;
414
+ if (onPartial) {
415
+ flatPromise
416
+ .then((flat) => {
417
+ if (settled || !flat.ok) return;
418
+ onPartial(flatResult(flat));
419
+ })
420
+ .catch(() => {});
421
+ }
238
422
 
239
423
  try {
240
424
  const payload = await getJson(`${rootOf(baseUrl)}/model/info`, headers, timeoutMs);
@@ -251,9 +435,11 @@ export async function fetchGatewayCatalog(baseUrl, apiKey, options = {}) {
251
435
  for (const key of Object.keys(typed)) {
252
436
  typed[key] = [...new Set(typed[key])].sort((a, b) => a.localeCompare(b));
253
437
  }
438
+ settled = true;
254
439
  return { ok: true, typed: true, source: 'model-info', ...typed };
255
440
  } catch (modelInfoError) {
256
- const flat = await fetchModels('ai-gateway', baseUrl, apiKey, { timeoutMs });
441
+ const flat = await flatPromise;
442
+ settled = true;
257
443
  if (!flat.ok) {
258
444
  return {
259
445
  ok: false,
@@ -265,19 +451,47 @@ export async function fetchGatewayCatalog(baseUrl, apiKey, options = {}) {
265
451
  error: flat.error,
266
452
  };
267
453
  }
268
- return {
269
- ok: true,
270
- typed: false,
271
- source: 'models',
272
- chat: flat.models,
273
- embedding: flat.models,
274
- rerank: flat.models,
275
- // Conservée pour l'affichage : elle explique pourquoi les listes ne sont
276
- // pas filtrées.
277
- error:
278
- modelInfoError instanceof Error ? modelInfoError.message : String(modelInfoError),
279
- };
454
+ return flatResult(flat, modelInfoError);
455
+ }
456
+ }
457
+
458
+ /**
459
+ * Catalogue d'un serveur unique, typé quand le serveur le permet.
460
+ *
461
+ * Même forme de retour que `fetchGatewayCatalog`, pour que le wizard n'ait
462
+ * qu'un seul objet à afficher. Un seul appel : chat et embeddings partagent
463
+ * l'endpoint, les interroger séparément revenait à poser deux fois la même
464
+ * question.
465
+ */
466
+ export async function fetchServerCatalog(provider, baseUrl, apiKey, options = {}) {
467
+ const engine = options.engine ?? provider;
468
+ const timeoutMs = options.timeoutMs ?? DISCOVERY_TIMEOUT_MS;
469
+ const flat = await fetchModels(provider, baseUrl, apiKey, { engine, timeoutMs, raw: true });
470
+ if (!flat.ok) {
471
+ return { ok: false, typed: false, source: 'unreachable', chat: [], embedding: [], rerank: [], error: flat.error };
472
+ }
473
+
474
+ const typed = { chat: [], embedding: [], rerank: [] };
475
+ for (const item of flat.items ?? []) {
476
+ const name = item?.id ?? item?.name ?? item?.model;
477
+ const kind = classifyModelEntry(item);
478
+ if (name && kind) typed[kind].push(String(name));
479
+ }
480
+ // Typage partiel accepté : un serveur peut n'annoncer que ses embeddings.
481
+ // Les listes vides retombent sur la liste complète plutôt que de rester
482
+ // vides — mieux vaut trop proposer que rien.
483
+ const classified = typed.chat.length + typed.embedding.length + typed.rerank.length;
484
+ if (classified === 0) {
485
+ return { ok: true, typed: false, source: 'models', chat: flat.models, embedding: flat.models, rerank: flat.models };
280
486
  }
487
+ return {
488
+ ok: true,
489
+ typed: true,
490
+ source: 'models',
491
+ chat: typed.chat.length ? sortedUnique(typed.chat) : flat.models,
492
+ embedding: typed.embedding.length ? sortedUnique(typed.embedding) : flat.models,
493
+ rerank: typed.rerank.length ? sortedUnique(typed.rerank) : flat.models,
494
+ };
281
495
  }
282
496
 
283
497
  export function fallbackModels(engine, kind) {
@@ -1,10 +1,15 @@
1
1
  import assert from 'node:assert/strict';
2
2
  import test from 'node:test';
3
3
  import {
4
+ DISCOVERY_TIMEOUT_MS,
5
+ classifyModelEntry,
6
+ describeFetchError,
7
+ fetchServerCatalog,
4
8
  fallbackModels,
5
9
  fetchGatewayCatalog,
6
10
  fetchModels,
7
11
  requiresBaseUrl,
12
+ transportSummary,
8
13
  } from './modelFetch.js';
9
14
 
10
15
  test('fetchModels returns remote OpenAI-compatible model ids', async () => {
@@ -105,6 +110,214 @@ test('fetchGatewayCatalog reports an unreachable gateway without inventing model
105
110
  }
106
111
  });
107
112
 
113
+ test('fetchGatewayCatalog queries /model/info and /v1/models in parallel', async () => {
114
+ const originalFetch = globalThis.fetch;
115
+ const started = [];
116
+ let releaseModelInfo;
117
+ const modelInfoGate = new Promise((resolve) => { releaseModelInfo = resolve; });
118
+ globalThis.fetch = async (url) => {
119
+ started.push(url);
120
+ if (url.endsWith('/model/info')) {
121
+ await modelInfoGate;
122
+ return { ok: false, status: 404 };
123
+ }
124
+ return { ok: true, json: async () => ({ data: [{ id: 'a' }] }) };
125
+ };
126
+ try {
127
+ const pending = fetchGatewayCatalog('http://gw:4000/v1', 'key', { timeoutMs: 100 });
128
+ // Le repli ne doit pas attendre la fin du chemin typé : les deux requêtes
129
+ // sont déjà parties quand /model/info est encore en vol.
130
+ await new Promise((resolve) => setImmediate(resolve));
131
+ assert.equal(started.length, 2);
132
+ releaseModelInfo();
133
+ const result = await pending;
134
+ assert.equal(result.ok, true);
135
+ assert.equal(result.typed, false);
136
+ assert.deepEqual(result.chat, ['a']);
137
+ } finally {
138
+ globalThis.fetch = originalFetch;
139
+ }
140
+ });
141
+
142
+ test('HTTP failures name the likely cause instead of a bare status', async () => {
143
+ const originalFetch = globalThis.fetch;
144
+ globalThis.fetch = async () => ({ ok: false, status: 401 });
145
+ try {
146
+ const result = await fetchModels('openai', 'https://api.openai.com/v1', 'bad', {
147
+ timeoutMs: 50,
148
+ });
149
+ assert.equal(result.ok, false);
150
+ assert.match(result.error, /HTTP 401/);
151
+ assert.match(result.error, /API key rejected/);
152
+ } finally {
153
+ globalThis.fetch = originalFetch;
154
+ }
155
+ });
156
+
157
+ test('describeFetchError turns transport codes into an actionable sentence', () => {
158
+ const tls = Object.assign(new Error('fetch failed'), {
159
+ cause: { code: 'SELF_SIGNED_CERT_IN_CHAIN' },
160
+ });
161
+ assert.match(describeFetchError(tls, { url: 'https://gw:4000/v1/models' }), /--cacert/);
162
+
163
+ const refused = Object.assign(new Error('fetch failed'), {
164
+ cause: { code: 'ECONNREFUSED' },
165
+ });
166
+ assert.match(describeFetchError(refused, { url: 'http://gw:4000/v1' }), /gw:4000/);
167
+
168
+ const dns = Object.assign(new Error('fetch failed'), { cause: { code: 'ENOTFOUND' } });
169
+ assert.match(describeFetchError(dns, { url: 'http://nope.local/v1' }), /host not found/);
170
+
171
+ // Le délai en millisecondes est un détail d'implémentation : le message
172
+ // nomme l'hôte muet et les causes probables, pas la valeur du timeout.
173
+ const aborted = Object.assign(new Error('aborted'), { name: 'AbortError' });
174
+ const timedOut = describeFetchError(aborted, { url: 'http://gw:4000/v1', timeoutMs: 8000 });
175
+ assert.match(timedOut, /gw:4000 did not answer in time/);
176
+ assert.match(timedOut, /proxy or firewall/);
177
+ assert.doesNotMatch(timedOut, /\d+ ms/);
178
+ });
179
+
180
+ test('transportSummary warns when a declared proxy is not actually used', () => {
181
+ assert.match(
182
+ transportSummary({ HTTPS_PROXY: 'http://proxy:3128' }),
183
+ /NODE_USE_ENV_PROXY not set/,
184
+ );
185
+ assert.match(
186
+ transportSummary({ HTTPS_PROXY: 'http://proxy:3128', NODE_USE_ENV_PROXY: '1' }),
187
+ /proxy http:\/\/proxy:3128/,
188
+ );
189
+ assert.match(transportSummary({}), /direct connection/);
190
+ assert.match(transportSummary({ WIKI_MANAGER_CACERT_PATH: '/ca.pem' }), /CA \/ca\.pem/);
191
+ });
192
+
193
+ test('the discovery timeout only bounds silent failures, never a healthy endpoint', () => {
194
+ // Assez long pour un endpoint distant lent, assez court pour qu'un proxy qui
195
+ // avale la connexion finisse par se dénoncer. Il n'est jamais attendu par
196
+ // une étape du wizard : la découverte tourne en tâche de fond.
197
+ assert.ok(DISCOVERY_TIMEOUT_MS >= 5000 && DISCOVERY_TIMEOUT_MS <= 15000);
198
+ });
199
+
200
+ test('the gateway hands over the flat list before /model/info completes', async () => {
201
+ const originalFetch = globalThis.fetch;
202
+ let releaseModelInfo;
203
+ const modelInfoGate = new Promise((resolve) => { releaseModelInfo = resolve; });
204
+ globalThis.fetch = async (url) => {
205
+ if (url.endsWith('/model/info')) {
206
+ await modelInfoGate;
207
+ return {
208
+ ok: true,
209
+ json: async () => ({
210
+ data: [{ model_name: 'chat-a', model_info: { mode: 'chat' } }],
211
+ }),
212
+ };
213
+ }
214
+ return { ok: true, json: async () => ({ data: [{ id: 'chat-a' }, { id: 'embed-b' }] }) };
215
+ };
216
+ try {
217
+ const partials = [];
218
+ const pending = fetchGatewayCatalog('http://gw:4000/v1', 'key', {
219
+ timeoutMs: 100,
220
+ onPartial: (partial) => partials.push(partial),
221
+ });
222
+ // La liste plate doit être livrée pendant que /model/info est encore en vol.
223
+ await new Promise((resolve) => setTimeout(resolve, 5));
224
+ assert.equal(partials.length, 1);
225
+ assert.equal(partials[0].typed, false);
226
+ assert.deepEqual(partials[0].chat, ['chat-a', 'embed-b']);
227
+
228
+ releaseModelInfo();
229
+ const result = await pending;
230
+ assert.equal(result.typed, true);
231
+ assert.deepEqual(result.chat, ['chat-a']);
232
+ // Le typage arrivé, plus aucune livraison partielle ne doit suivre.
233
+ assert.equal(partials.length, 1);
234
+ } finally {
235
+ globalThis.fetch = originalFetch;
236
+ }
237
+ });
238
+
239
+ test('classifyModelEntry reads the type hints real servers actually send', () => {
240
+ // Albert.
241
+ assert.equal(classifyModelEntry({ type: 'text-generation' }), 'chat');
242
+ assert.equal(classifyModelEntry({ type: 'text-embeddings-inference' }), 'embedding');
243
+ assert.equal(classifyModelEntry({ type: 'text-classification' }), 'rerank');
244
+ // Hors périmètre : absent des trois listes plutôt que mal classé.
245
+ assert.equal(classifyModelEntry({ type: 'automatic-speech-recognition' }), null);
246
+ // LiteLLM.
247
+ assert.equal(classifyModelEntry({ model_info: { mode: 'embedding' } }), 'embedding');
248
+ // OpenAI ne type rien : `object: "model"` ne doit pas être pris pour un type.
249
+ assert.equal(classifyModelEntry({ id: 'gpt-4.1', object: 'model' }), null);
250
+ });
251
+
252
+ test('fetchServerCatalog splits a direct server catalog by model type', async () => {
253
+ const originalFetch = globalThis.fetch;
254
+ globalThis.fetch = async () => ({
255
+ ok: true,
256
+ json: async () => ({
257
+ data: [
258
+ { id: 'openai/gpt-oss-120b', type: 'text-generation' },
259
+ { id: 'bge-m3', type: 'text-embeddings-inference' },
260
+ { id: 'bge-reranker-v2-m3', type: 'text-classification' },
261
+ { id: 'whisper-large-v3', type: 'automatic-speech-recognition' },
262
+ ],
263
+ }),
264
+ });
265
+ try {
266
+ const result = await fetchServerCatalog('openai-compatible', 'https://albert.example/v1', 'key', {
267
+ engine: 'albert',
268
+ timeoutMs: 100,
269
+ });
270
+ assert.equal(result.ok, true);
271
+ assert.equal(result.typed, true);
272
+ assert.deepEqual(result.chat, ['openai/gpt-oss-120b']);
273
+ assert.deepEqual(result.embedding, ['bge-m3']);
274
+ assert.deepEqual(result.rerank, ['bge-reranker-v2-m3']);
275
+ } finally {
276
+ globalThis.fetch = originalFetch;
277
+ }
278
+ });
279
+
280
+ test('fetchServerCatalog stays untyped when the server types nothing', async () => {
281
+ const originalFetch = globalThis.fetch;
282
+ globalThis.fetch = async () => ({
283
+ ok: true,
284
+ json: async () => ({ data: [{ id: 'b', object: 'model' }, { id: 'a', object: 'model' }] }),
285
+ });
286
+ try {
287
+ const result = await fetchServerCatalog('openai-compatible', 'https://api.openai.com/v1', 'key', {
288
+ engine: 'openai',
289
+ timeoutMs: 100,
290
+ });
291
+ assert.equal(result.typed, false);
292
+ assert.deepEqual(result.chat, ['a', 'b']);
293
+ assert.deepEqual(result.embedding, ['a', 'b']);
294
+ } finally {
295
+ globalThis.fetch = originalFetch;
296
+ }
297
+ });
298
+
299
+ test('a partially typed catalog falls back to the full list, never to an empty one', async () => {
300
+ const originalFetch = globalThis.fetch;
301
+ globalThis.fetch = async () => ({
302
+ ok: true,
303
+ json: async () => ({
304
+ data: [{ id: 'embed-only', type: 'text-embeddings-inference' }, { id: 'mystery' }],
305
+ }),
306
+ });
307
+ try {
308
+ const result = await fetchServerCatalog('openai-compatible', 'https://x/v1', 'key', {
309
+ timeoutMs: 100,
310
+ });
311
+ assert.equal(result.typed, true);
312
+ assert.deepEqual(result.embedding, ['embed-only']);
313
+ // Aucun modèle annoncé comme chat : proposer toute la liste vaut mieux
314
+ // qu'une étape sans aucune suggestion.
315
+ assert.deepEqual(result.chat, ['embed-only', 'mystery']);
316
+ } finally {
317
+ globalThis.fetch = originalFetch;
318
+ }
319
+ });
320
+
108
321
  test('requiresBaseUrl follows the engine, and the gateway always needs one', () => {
109
322
  assert.equal(requiresBaseUrl('openai-compatible', 'ollama'), true);
110
323
  assert.equal(requiresBaseUrl('openai-compatible', 'openai'), false);