@dotdrelle/wiki-manager 0.15.32 → 0.15.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +8 -5
- package/agents.docker-compose.yml +11 -16
- package/docker-compose.yml +9 -8
- package/package.json +2 -2
- package/src/agent/graph.js +20 -0
- package/src/commands/slash.js +73 -10
- package/src/commands/slash.test.js +37 -3
- package/src/core/agentsCompose.js +86 -1
- package/src/core/buildInfo.json +2 -2
- package/src/core/compose.js +24 -2
- package/src/core/dockerCompose.test.js +18 -0
- package/src/core/googleGrants.js +38 -0
- package/src/core/googleGrants.test.js +59 -0
- package/src/core/mcp.js +1 -1
- package/src/core/modelFetch.js +449 -47
- package/src/core/modelFetch.test.js +290 -2
- package/src/core/profileServiceStatus.test.js +146 -0
- package/src/core/wikiSetup.js +43 -4
- package/src/core/wikiWorkspace.test.js +9 -3
- package/src/core/wikirc.test.js +59 -5
- package/src/runtime/lifecycle.js +105 -3
- package/src/runtime/lifecycle.test.js +68 -0
- package/src/shell/SetupWizard.tsx +617 -66
- package/src/shell/repl.js +37 -11
- package/src/shell/setupWizardDiscovery.test.js +48 -0
- package/src/shell/setupWizardPlaceholders.test.js +16 -3
- package/src/shell/setupWizardSuggestions.test.js +58 -0
- package/src/shell/wrapText.js +57 -0
- package/src/shell/wrapText.test.js +48 -0
- package/wiki-workspace +116 -32
package/src/core/mcp.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { existsSync, readFileSync } from 'node:fs';
|
|
2
2
|
import { managerEnvFile, managerMcpEndpointsFile, readEnvFile } from './env.js';
|
|
3
3
|
|
|
4
|
-
const WIKI_MANAGER_VERSION = '0.15.
|
|
4
|
+
const WIKI_MANAGER_VERSION = '0.15.35';
|
|
5
5
|
|
|
6
6
|
function envValue(key) {
|
|
7
7
|
const filePath = managerEnvFile();
|
package/src/core/modelFetch.js
CHANGED
|
@@ -1,97 +1,499 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Découverte des modèles disponibles, pour alimenter le wizard.
|
|
3
|
+
*
|
|
4
|
+
* Deux chemins, correspondant aux deux valeurs de `llm.provider` :
|
|
5
|
+
*
|
|
6
|
+
* - `openai-compatible` : un serveur unique. L'endpoint et les en-têtes
|
|
7
|
+
* dépendent du moteur (`engine`), d'où les tables ci-dessous.
|
|
8
|
+
* - `ai-gateway` : un seul chemin, `GET /v1/models`, plus `GET /model/info`
|
|
9
|
+
* quand il est disponible — c'est lui qui porte le type de chaque modèle
|
|
10
|
+
* (chat, embedding, rerank) et permet de filtrer les listes du wizard.
|
|
11
|
+
*/
|
|
12
|
+
|
|
1
13
|
const FALLBACK_MODELS = {
|
|
2
14
|
openai: ['gpt-5.4', 'gpt-5.4-mini', 'gpt-4.1', 'gpt-4.1-mini'],
|
|
3
15
|
anthropic: ['claude-sonnet-4-5', 'claude-opus-4-1', 'claude-3-7-sonnet-latest'],
|
|
4
16
|
ollama: ['llama3.2', 'qwen2.5', 'mistral', 'nomic-embed-text'],
|
|
5
|
-
|
|
6
|
-
|
|
17
|
+
vllm: ['Qwen/Qwen2.5-7B-Instruct', 'meta-llama/Llama-3.1-8B-Instruct'],
|
|
18
|
+
mlx: ['mlx-community/Qwen2.5-7B-Instruct-4bit'],
|
|
19
|
+
albert: ['albert-large', 'albert-small'],
|
|
20
|
+
generic: ['gpt-4.1-mini', 'llama3.2'],
|
|
7
21
|
};
|
|
8
22
|
|
|
9
23
|
const FALLBACK_EMBEDDINGS = {
|
|
10
24
|
openai: ['text-embedding-3-small', 'text-embedding-3-large'],
|
|
11
25
|
anthropic: ['text-embedding-3-small'],
|
|
12
26
|
ollama: ['nomic-embed-text', 'mxbai-embed-large'],
|
|
13
|
-
|
|
14
|
-
|
|
27
|
+
vllm: ['BAAI/bge-m3'],
|
|
28
|
+
mlx: ['BAAI/bge-m3'],
|
|
29
|
+
albert: ['BAAI/bge-m3'],
|
|
30
|
+
generic: ['BAAI/bge-m3', 'text-embedding-3-small', 'nomic-embed-text'],
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
export const PROVIDERS = ['openai-compatible', 'ai-gateway'];
|
|
34
|
+
|
|
35
|
+
export const ENGINES = [
|
|
36
|
+
'ollama',
|
|
37
|
+
'vllm',
|
|
38
|
+
'mlx',
|
|
39
|
+
'albert',
|
|
40
|
+
'openai',
|
|
41
|
+
'anthropic',
|
|
42
|
+
'generic',
|
|
43
|
+
];
|
|
44
|
+
|
|
45
|
+
/** Moteurs qui exigent une baseUrl explicite — il n'existe pas de défaut sensé. */
|
|
46
|
+
const ENGINES_REQUIRING_BASE_URL = new Set(['ollama', 'vllm', 'mlx', 'generic']);
|
|
47
|
+
|
|
48
|
+
const ENGINE_DEFAULT_BASE_URL = {
|
|
49
|
+
openai: 'https://api.openai.com/v1',
|
|
50
|
+
anthropic: 'https://api.anthropic.com/v1',
|
|
51
|
+
albert: 'https://albert.api.etalab.gouv.fr/v1',
|
|
52
|
+
ollama: 'http://127.0.0.1:11434/v1',
|
|
53
|
+
vllm: 'http://127.0.0.1:8000/v1',
|
|
54
|
+
mlx: 'http://127.0.0.1:8080/v1',
|
|
15
55
|
};
|
|
16
56
|
|
|
57
|
+
export function requiresBaseUrl(provider, engine) {
|
|
58
|
+
if (normalizeProvider(provider) === 'ai-gateway') return true;
|
|
59
|
+
return ENGINES_REQUIRING_BASE_URL.has(normalizeEngine(engine));
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function defaultBaseUrl(provider, engine) {
|
|
63
|
+
if (normalizeProvider(provider) === 'ai-gateway') return '';
|
|
64
|
+
return ENGINE_DEFAULT_BASE_URL[normalizeEngine(engine)] ?? '';
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/** Routage. Tolérant aux libellés du wizard. */
|
|
17
68
|
export function normalizeProvider(provider) {
|
|
18
69
|
const value = String(provider ?? '').toLowerCase();
|
|
19
|
-
if (value.includes('
|
|
20
|
-
if (value.includes('anthropic')) return 'anthropic';
|
|
21
|
-
if (value.includes('ollama')) return 'ollama';
|
|
22
|
-
if (value.includes('openai')) return 'openai';
|
|
70
|
+
if (value.includes('gateway')) return 'ai-gateway';
|
|
23
71
|
return 'openai-compatible';
|
|
24
72
|
}
|
|
25
73
|
|
|
26
|
-
|
|
27
|
-
|
|
74
|
+
/**
|
|
75
|
+
* Libellés du wizard vers moteur. Correspondance **exacte**, pas par sous-chaîne.
|
|
76
|
+
*
|
|
77
|
+
* Une recherche par sous-chaîne était fausse : « Other (generic
|
|
78
|
+
* OpenAI-compatible) » contient « openai », qui était testé avant « generic »
|
|
79
|
+
* — l'option « serveur générique » persistait donc `engine: openai`, avec les
|
|
80
|
+
* contournements inversés. Et aucun libellé ne pouvait plus résoudre vers
|
|
81
|
+
* `generic`, ce qui cassait la présélection à la réouverture du wizard.
|
|
82
|
+
*/
|
|
83
|
+
const ENGINE_LABELS = new Map([
|
|
84
|
+
['openai', 'openai'],
|
|
85
|
+
['anthropic', 'anthropic'],
|
|
86
|
+
['ollama (local)', 'ollama'],
|
|
87
|
+
['vllm (local)', 'vllm'],
|
|
88
|
+
['mlx (local)', 'mlx'],
|
|
89
|
+
['albert', 'albert'],
|
|
90
|
+
['other (generic openai-compatible)', 'generic'],
|
|
91
|
+
]);
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* Moteur. Accepte les libellés du wizard, les valeurs canoniques, et les
|
|
95
|
+
* anciennes valeurs de `provider` (`openai`, `ollama`, `anthropic`) devenues
|
|
96
|
+
* des moteurs.
|
|
97
|
+
*/
|
|
98
|
+
export function normalizeEngine(engine) {
|
|
99
|
+
const value = String(engine ?? '').trim().toLowerCase();
|
|
100
|
+
const fromLabel = ENGINE_LABELS.get(value);
|
|
101
|
+
if (fromLabel) return fromLabel;
|
|
102
|
+
if (ENGINES.includes(value)) return value;
|
|
103
|
+
// Repli tolérant, utile pour les valeurs libres ; l'ordre importe donc les
|
|
104
|
+
// moteurs les plus spécifiques passent avant les plus génériques.
|
|
105
|
+
for (const candidate of ENGINES) {
|
|
106
|
+
if (candidate !== 'generic' && value.includes(candidate)) return candidate;
|
|
107
|
+
}
|
|
108
|
+
return 'generic';
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
function fallbackFor(engine, kind) {
|
|
112
|
+
const normalized = normalizeEngine(engine);
|
|
28
113
|
const source = kind === 'embedding' ? FALLBACK_EMBEDDINGS : FALLBACK_MODELS;
|
|
29
|
-
return source[normalized] ?? source.
|
|
114
|
+
return source[normalized] ?? source.generic;
|
|
30
115
|
}
|
|
31
116
|
|
|
32
|
-
function
|
|
33
|
-
|
|
117
|
+
function trimUrl(url) {
|
|
118
|
+
return String(url ?? '').replace(/\/+$/g, '');
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* `baseUrl` est écrite avec son suffixe `/v1` dans le wikirc. Les endpoints de
|
|
123
|
+
* listing vivent tantôt sous `/v1` (OpenAI), tantôt à la racine (Ollama,
|
|
124
|
+
* `/model/info` de LiteLLM) — d'où cette racine sans suffixe.
|
|
125
|
+
*/
|
|
126
|
+
function rootOf(baseUrl) {
|
|
127
|
+
return trimUrl(baseUrl).replace(/\/v1$/, '');
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
function endpointFor(provider, engine, baseUrl) {
|
|
131
|
+
if (normalizeProvider(provider) === 'ai-gateway') {
|
|
132
|
+
return `${rootOf(baseUrl)}/v1/models`;
|
|
133
|
+
}
|
|
134
|
+
const normalized = normalizeEngine(engine);
|
|
34
135
|
if (normalized === 'anthropic') return 'https://api.anthropic.com/v1/models';
|
|
35
|
-
const root =
|
|
136
|
+
const root = rootOf(baseUrl) || 'https://api.openai.com';
|
|
36
137
|
return normalized === 'ollama' ? `${root}/api/tags` : `${root}/v1/models`;
|
|
37
138
|
}
|
|
38
139
|
|
|
39
|
-
function headersFor(provider, apiKey) {
|
|
40
|
-
|
|
140
|
+
function headersFor(provider, engine, apiKey) {
|
|
141
|
+
if (normalizeProvider(provider) === 'ai-gateway') {
|
|
142
|
+
return { Authorization: `Bearer ${apiKey}` };
|
|
143
|
+
}
|
|
144
|
+
const normalized = normalizeEngine(engine);
|
|
41
145
|
if (normalized === 'ollama') return {};
|
|
42
146
|
if (normalized === 'anthropic') {
|
|
43
|
-
return {
|
|
44
|
-
'x-api-key': apiKey,
|
|
45
|
-
'anthropic-version': '2023-06-01',
|
|
46
|
-
};
|
|
147
|
+
return { 'x-api-key': apiKey, 'anthropic-version': '2023-06-01' };
|
|
47
148
|
}
|
|
48
149
|
return { Authorization: `Bearer ${apiKey}` };
|
|
49
150
|
}
|
|
50
151
|
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
152
|
+
/**
|
|
153
|
+
* Type d'un modèle, quand la réponse le porte.
|
|
154
|
+
*
|
|
155
|
+
* `/v1/models` d'OpenAI ne type rien (`object: "model"` partout) — d'où le
|
|
156
|
+
* repli non typé. Mais plusieurs serveurs OpenAI-compatibles ajoutent un
|
|
157
|
+
* champ : Albert annonce `text-generation`, `text-embeddings-inference` ou
|
|
158
|
+
* `text-classification` (son reranker), LiteLLM porte `model_info.mode`. Les
|
|
159
|
+
* ignorer forçait le wizard à proposer les modèles de chat pour l'étape
|
|
160
|
+
* embeddings — sur Albert, aucune suggestion ne pouvait correspondre.
|
|
161
|
+
*
|
|
162
|
+
* Un indice non reconnu (`automatic-speech-recognition`) rend `null` : le
|
|
163
|
+
* modèle est simplement absent des trois listes.
|
|
164
|
+
*/
|
|
165
|
+
export function classifyModelEntry(item) {
|
|
166
|
+
const hints = [
|
|
167
|
+
item?.model_info?.mode,
|
|
168
|
+
item?.mode,
|
|
169
|
+
item?.type,
|
|
170
|
+
item?.task,
|
|
171
|
+
item?.object,
|
|
172
|
+
...(Array.isArray(item?.capabilities) ? item.capabilities : []),
|
|
173
|
+
]
|
|
174
|
+
.filter((value) => typeof value === 'string')
|
|
175
|
+
.map((value) => value.toLowerCase());
|
|
176
|
+
|
|
177
|
+
for (const hint of hints) {
|
|
178
|
+
if (hint.includes('embed')) return 'embedding';
|
|
179
|
+
// Albert expose son reranker en `text-classification`.
|
|
180
|
+
if (hint.includes('rerank') || hint.includes('classification')) return 'rerank';
|
|
181
|
+
if (hint.includes('chat') || hint.includes('generation') || hint.includes('completion')) {
|
|
182
|
+
return 'chat';
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
return null;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
function itemsOf(provider, engine, payload) {
|
|
189
|
+
return normalizeProvider(provider) === 'openai-compatible' && normalizeEngine(engine) === 'ollama'
|
|
190
|
+
? payload?.models
|
|
191
|
+
: payload?.data;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
function sortedUnique(values) {
|
|
195
|
+
return [...new Set(values)].sort((a, b) => a.localeCompare(b));
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
function parseModelNames(provider, engine, payload) {
|
|
199
|
+
const items = itemsOf(provider, engine, payload);
|
|
54
200
|
if (!Array.isArray(items)) return [];
|
|
55
|
-
return
|
|
56
|
-
.map((item) => item?.id ?? item?.name ?? item?.model)
|
|
57
|
-
|
|
58
|
-
.map(String)
|
|
59
|
-
.sort((a, b) => a.localeCompare(b));
|
|
201
|
+
return sortedUnique(
|
|
202
|
+
items.map((item) => item?.id ?? item?.name ?? item?.model).filter(Boolean).map(String),
|
|
203
|
+
);
|
|
60
204
|
}
|
|
61
205
|
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
206
|
+
/**
|
|
207
|
+
* Délai de découverte du wizard.
|
|
208
|
+
*
|
|
209
|
+
* Un `/v1/models` qui répond le fait en quelques dizaines de millisecondes :
|
|
210
|
+
* ce délai n'est jamais payé par un endpoint sain, il ne borne que les pannes
|
|
211
|
+
* silencieuses (proxy qui avale la connexion, port filtré). Il peut donc
|
|
212
|
+
* rester confortable — la découverte est lancée en tâche de fond, aucune
|
|
213
|
+
* étape du wizard ne l'attend.
|
|
214
|
+
*/
|
|
215
|
+
export const DISCOVERY_TIMEOUT_MS = 8000;
|
|
216
|
+
|
|
217
|
+
/** Codes TLS qui désignent une CA privée ou un proxy qui intercepte. */
|
|
218
|
+
const TLS_ERROR_CODES = new Set([
|
|
219
|
+
'UNABLE_TO_VERIFY_LEAF_SIGNATURE',
|
|
220
|
+
'SELF_SIGNED_CERT_IN_CHAIN',
|
|
221
|
+
'DEPTH_ZERO_SELF_SIGNED_CERT',
|
|
222
|
+
'CERT_HAS_EXPIRED',
|
|
223
|
+
'CERT_UNTRUSTED',
|
|
224
|
+
'ERR_TLS_CERT_ALTNAME_INVALID',
|
|
225
|
+
'UNABLE_TO_GET_ISSUER_CERT_LOCALLY',
|
|
226
|
+
]);
|
|
227
|
+
|
|
228
|
+
function hostOf(url) {
|
|
229
|
+
try {
|
|
230
|
+
return new URL(url).host;
|
|
231
|
+
} catch {
|
|
232
|
+
return url;
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
function httpStatusHint(status, url) {
|
|
237
|
+
if (status === 401) return `HTTP 401 — API key rejected by ${hostOf(url)}`;
|
|
238
|
+
if (status === 403) {
|
|
239
|
+
return `HTTP 403 — key accepted but access to the model catalog is denied`;
|
|
240
|
+
}
|
|
241
|
+
if (status === 404) return `HTTP 404 — no model catalog exposed at ${url}`;
|
|
242
|
+
if (status === 407) {
|
|
243
|
+
return `HTTP 407 — the HTTP proxy requires authentication (check HTTPS_PROXY credentials)`;
|
|
244
|
+
}
|
|
245
|
+
if (status >= 500) return `HTTP ${status} — the server failed to answer ${url}`;
|
|
246
|
+
return `HTTP ${status} on ${url}`;
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
/**
|
|
250
|
+
* Message actionnable pour un échec réseau.
|
|
251
|
+
*
|
|
252
|
+
* Le message brut de `fetch` ("fetch failed") ne dit rien : la cause utile est
|
|
253
|
+
* dans `err.cause.code`. On la traduit en une phrase qui nomme la manœuvre —
|
|
254
|
+
* proxy, CA privée, port fermé, DNS — parce que c'est exactement ce que
|
|
255
|
+
* l'opérateur doit corriger, et qu'il ne le devinera pas depuis le wizard.
|
|
256
|
+
*/
|
|
257
|
+
export function describeFetchError(err, { url, timeoutMs } = {}) {
|
|
258
|
+
if (!err) return 'unknown error';
|
|
259
|
+
if (err.name === 'AbortError' || err.name === 'TimeoutError') {
|
|
260
|
+
// Le délai en millisecondes est un détail d'implémentation : ce qui aide
|
|
261
|
+
// l'opérateur, c'est l'hôte qui n'a pas répondu et les causes probables.
|
|
262
|
+
return `${hostOf(url)} did not answer in time — server unreachable, or blocked by a proxy or firewall`;
|
|
263
|
+
}
|
|
264
|
+
const code = err?.cause?.code ?? err?.code ?? null;
|
|
265
|
+
if (code && TLS_ERROR_CODES.has(code)) {
|
|
266
|
+
return `TLS certificate rejected (${code}) — private CA or intercepting proxy; relaunch with wiki-manager --cacert <file.pem>`;
|
|
66
267
|
}
|
|
67
|
-
|
|
268
|
+
if (code === 'ECONNREFUSED') {
|
|
269
|
+
return `connection refused (ECONNREFUSED) by ${hostOf(url)} — nothing is listening on this host/port`;
|
|
270
|
+
}
|
|
271
|
+
if (code === 'ENOTFOUND' || code === 'EAI_AGAIN') {
|
|
272
|
+
return `host not found (${code}): ${hostOf(url)} — check the URL, DNS, or that the proxy resolves it`;
|
|
273
|
+
}
|
|
274
|
+
if (code === 'ETIMEDOUT') {
|
|
275
|
+
return `connection timed out (ETIMEDOUT) to ${hostOf(url)} — usually a firewall dropping the packets`;
|
|
276
|
+
}
|
|
277
|
+
if (code === 'ECONNRESET' || code === 'EPROTO') {
|
|
278
|
+
return `connection reset (${code}) by ${hostOf(url)} — often a proxy intercepting TLS, or http:// used on an https:// endpoint`;
|
|
279
|
+
}
|
|
280
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
281
|
+
return code ? `${message} (${code})` : message;
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* État du transport local, affiché à côté d'une erreur de découverte.
|
|
286
|
+
*
|
|
287
|
+
* Un proxy déclaré mais non activé (`NODE_USE_ENV_PROXY` absent) est la panne
|
|
288
|
+
* la plus fréquente en entreprise, et elle est invisible sans ce rappel.
|
|
289
|
+
*/
|
|
290
|
+
export function transportSummary(env = process.env) {
|
|
291
|
+
const proxy = env.HTTPS_PROXY ?? env.HTTP_PROXY ?? null;
|
|
292
|
+
const parts = [];
|
|
293
|
+
if (proxy) {
|
|
294
|
+
parts.push(
|
|
295
|
+
env.NODE_USE_ENV_PROXY === '1'
|
|
296
|
+
? `proxy ${proxy}`
|
|
297
|
+
: `proxy ${proxy} (NODE_USE_ENV_PROXY not set: it is NOT used)`,
|
|
298
|
+
);
|
|
299
|
+
} else {
|
|
300
|
+
parts.push('direct connection (no HTTP(S)_PROXY)');
|
|
301
|
+
}
|
|
302
|
+
const cacert = env.WIKI_MANAGER_CACERT_PATH ?? env.NODE_EXTRA_CA_CERTS ?? null;
|
|
303
|
+
parts.push(cacert ? `CA ${cacert}` : 'CA system trust store');
|
|
304
|
+
return parts.join(' · ');
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
async function getJson(url, headers, timeoutMs) {
|
|
68
308
|
const controller = new AbortController();
|
|
69
309
|
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
|
70
310
|
try {
|
|
71
|
-
|
|
311
|
+
const response = await fetch(url, { headers, signal: controller.signal });
|
|
312
|
+
if (!response.ok) throw new Error(httpStatusHint(response.status, url));
|
|
313
|
+
return await response.json();
|
|
314
|
+
} catch (err) {
|
|
315
|
+
if (err instanceof Error && /^HTTP \d/.test(err.message)) throw err;
|
|
316
|
+
throw new Error(describeFetchError(err, { url, timeoutMs }));
|
|
317
|
+
} finally {
|
|
318
|
+
clearTimeout(timer);
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/**
|
|
323
|
+
* Liste plate des modèles.
|
|
324
|
+
*
|
|
325
|
+
* `options.engine` porte le moteur ; à défaut, le premier argument est
|
|
326
|
+
* réinterprété comme tel, ce qui garde les appels historiques valides.
|
|
327
|
+
*/
|
|
328
|
+
export async function fetchModels(provider, baseUrl, apiKey, options = {}) {
|
|
329
|
+
const routing = normalizeProvider(provider);
|
|
330
|
+
const normalizedEngine = normalizeEngine(options.engine ?? provider);
|
|
331
|
+
|
|
332
|
+
if (routing === 'openai-compatible' && normalizedEngine === 'anthropic') {
|
|
333
|
+
return {
|
|
334
|
+
ok: false,
|
|
335
|
+
models: fallbackFor(normalizedEngine, options.kind),
|
|
336
|
+
source: 'fallback',
|
|
337
|
+
error: 'Anthropic model listing is not supported',
|
|
338
|
+
};
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
const timeoutMs = options.timeoutMs ?? DISCOVERY_TIMEOUT_MS;
|
|
342
|
+
try {
|
|
343
|
+
const needsKey = !(routing === 'openai-compatible' && normalizedEngine === 'ollama');
|
|
344
|
+
if (needsKey && !apiKey) {
|
|
72
345
|
throw new Error('API key is required to fetch remote models');
|
|
73
346
|
}
|
|
74
|
-
const
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
const
|
|
80
|
-
const models = parseModelNames(normalized, payload);
|
|
347
|
+
const payload = await getJson(
|
|
348
|
+
endpointFor(provider, normalizedEngine, baseUrl),
|
|
349
|
+
headersFor(provider, normalizedEngine, apiKey),
|
|
350
|
+
timeoutMs,
|
|
351
|
+
);
|
|
352
|
+
const models = parseModelNames(provider, normalizedEngine, payload);
|
|
81
353
|
if (models.length === 0) throw new Error('No models returned');
|
|
82
|
-
|
|
354
|
+
// `raw` rend les entrées brutes, seules porteuses des indices de type que
|
|
355
|
+
// `fetchServerCatalog` exploite. Absentes par défaut : la forme historique
|
|
356
|
+
// de ce retour est {ok, models, source}.
|
|
357
|
+
return options.raw
|
|
358
|
+
? { ok: true, models, source: 'remote', items: itemsOf(provider, normalizedEngine, payload) ?? [] }
|
|
359
|
+
: { ok: true, models, source: 'remote' };
|
|
83
360
|
} catch (err) {
|
|
84
361
|
return {
|
|
85
362
|
ok: false,
|
|
86
|
-
models: fallbackFor(
|
|
363
|
+
models: fallbackFor(normalizedEngine, options.kind),
|
|
87
364
|
source: 'fallback',
|
|
88
365
|
error: err instanceof Error ? err.message : String(err),
|
|
89
366
|
};
|
|
90
|
-
} finally {
|
|
91
|
-
clearTimeout(timer);
|
|
92
367
|
}
|
|
93
368
|
}
|
|
94
369
|
|
|
95
|
-
|
|
96
|
-
|
|
370
|
+
/**
|
|
371
|
+
* Catalogue typé d'une gateway.
|
|
372
|
+
*
|
|
373
|
+
* Dégradation gracieuse en trois temps — jamais un catch silencieux vers un
|
|
374
|
+
* défaut :
|
|
375
|
+
*
|
|
376
|
+
* 1. `GET /model/info` porte `model_info.mode` : on sait quel modèle est un
|
|
377
|
+
* chat, un embedding ou un reranker, et le wizard filtre ses listes.
|
|
378
|
+
* 2. `GET /v1/models` ne renvoie qu'une liste plate : les trois listes
|
|
379
|
+
* reçoivent la même chose, et `typed: false` permet à l'appelant de le
|
|
380
|
+
* dire à l'utilisateur.
|
|
381
|
+
* 3. Injoignable : listes vides, `error` renseignée. Le wizard garde sa
|
|
382
|
+
* saisie libre, qui fait foi de toute façon.
|
|
383
|
+
*
|
|
384
|
+
* Les deux appels partent **en parallèle**, et le résultat est livré en deux
|
|
385
|
+
* temps : `options.onPartial` reçoit la liste plate de `/v1/models` dès
|
|
386
|
+
* qu'elle arrive — c'est la requête rapide, et elle suffit à choisir un
|
|
387
|
+
* modèle — pendant que `/model/info`, plus lourd côté gateway, continue.
|
|
388
|
+
* La promesse résout ensuite avec le catalogue typé s'il aboutit. L'opérateur
|
|
389
|
+
* a donc une liste utilisable immédiatement, qui se raffine sous ses yeux.
|
|
390
|
+
*/
|
|
391
|
+
export async function fetchGatewayCatalog(baseUrl, apiKey, options = {}) {
|
|
392
|
+
const timeoutMs = options.timeoutMs ?? DISCOVERY_TIMEOUT_MS;
|
|
393
|
+
const headers = { Authorization: `Bearer ${apiKey}` };
|
|
394
|
+
const onPartial = typeof options.onPartial === 'function' ? options.onPartial : null;
|
|
395
|
+
|
|
396
|
+
const flatPromise = fetchModels('ai-gateway', baseUrl, apiKey, { timeoutMs });
|
|
397
|
+
// Sans ce no-op, un rejet arrivant avant son `await` remonterait en
|
|
398
|
+
// unhandledRejection quand le chemin typé réussit.
|
|
399
|
+
flatPromise.catch(() => {});
|
|
400
|
+
|
|
401
|
+
const flatResult = (flat, error) => ({
|
|
402
|
+
ok: true,
|
|
403
|
+
typed: false,
|
|
404
|
+
source: 'models',
|
|
405
|
+
chat: flat.models,
|
|
406
|
+
embedding: flat.models,
|
|
407
|
+
rerank: flat.models,
|
|
408
|
+
// Conservée pour l'affichage : elle explique pourquoi les listes ne sont
|
|
409
|
+
// pas filtrées.
|
|
410
|
+
...(error ? { error: error instanceof Error ? error.message : String(error) } : {}),
|
|
411
|
+
});
|
|
412
|
+
|
|
413
|
+
let settled = false;
|
|
414
|
+
if (onPartial) {
|
|
415
|
+
flatPromise
|
|
416
|
+
.then((flat) => {
|
|
417
|
+
if (settled || !flat.ok) return;
|
|
418
|
+
onPartial(flatResult(flat));
|
|
419
|
+
})
|
|
420
|
+
.catch(() => {});
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
try {
|
|
424
|
+
const payload = await getJson(`${rootOf(baseUrl)}/model/info`, headers, timeoutMs);
|
|
425
|
+
const items = Array.isArray(payload?.data) ? payload.data : [];
|
|
426
|
+
const typed = { chat: [], embedding: [], rerank: [] };
|
|
427
|
+
for (const item of items) {
|
|
428
|
+
const name = item?.model_name ?? item?.id ?? item?.model_info?.id;
|
|
429
|
+
const mode = item?.model_info?.mode;
|
|
430
|
+
if (!name || !mode || !(mode in typed)) continue;
|
|
431
|
+
typed[mode].push(String(name));
|
|
432
|
+
}
|
|
433
|
+
const total = typed.chat.length + typed.embedding.length + typed.rerank.length;
|
|
434
|
+
if (total === 0) throw new Error('No typed models returned by /model/info');
|
|
435
|
+
for (const key of Object.keys(typed)) {
|
|
436
|
+
typed[key] = [...new Set(typed[key])].sort((a, b) => a.localeCompare(b));
|
|
437
|
+
}
|
|
438
|
+
settled = true;
|
|
439
|
+
return { ok: true, typed: true, source: 'model-info', ...typed };
|
|
440
|
+
} catch (modelInfoError) {
|
|
441
|
+
const flat = await flatPromise;
|
|
442
|
+
settled = true;
|
|
443
|
+
if (!flat.ok) {
|
|
444
|
+
return {
|
|
445
|
+
ok: false,
|
|
446
|
+
typed: false,
|
|
447
|
+
source: 'unreachable',
|
|
448
|
+
chat: [],
|
|
449
|
+
embedding: [],
|
|
450
|
+
rerank: [],
|
|
451
|
+
error: flat.error,
|
|
452
|
+
};
|
|
453
|
+
}
|
|
454
|
+
return flatResult(flat, modelInfoError);
|
|
455
|
+
}
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
/**
|
|
459
|
+
* Catalogue d'un serveur unique, typé quand le serveur le permet.
|
|
460
|
+
*
|
|
461
|
+
* Même forme de retour que `fetchGatewayCatalog`, pour que le wizard n'ait
|
|
462
|
+
* qu'un seul objet à afficher. Un seul appel : chat et embeddings partagent
|
|
463
|
+
* l'endpoint, les interroger séparément revenait à poser deux fois la même
|
|
464
|
+
* question.
|
|
465
|
+
*/
|
|
466
|
+
export async function fetchServerCatalog(provider, baseUrl, apiKey, options = {}) {
|
|
467
|
+
const engine = options.engine ?? provider;
|
|
468
|
+
const timeoutMs = options.timeoutMs ?? DISCOVERY_TIMEOUT_MS;
|
|
469
|
+
const flat = await fetchModels(provider, baseUrl, apiKey, { engine, timeoutMs, raw: true });
|
|
470
|
+
if (!flat.ok) {
|
|
471
|
+
return { ok: false, typed: false, source: 'unreachable', chat: [], embedding: [], rerank: [], error: flat.error };
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
const typed = { chat: [], embedding: [], rerank: [] };
|
|
475
|
+
for (const item of flat.items ?? []) {
|
|
476
|
+
const name = item?.id ?? item?.name ?? item?.model;
|
|
477
|
+
const kind = classifyModelEntry(item);
|
|
478
|
+
if (name && kind) typed[kind].push(String(name));
|
|
479
|
+
}
|
|
480
|
+
// Typage partiel accepté : un serveur peut n'annoncer que ses embeddings.
|
|
481
|
+
// Les listes vides retombent sur la liste complète plutôt que de rester
|
|
482
|
+
// vides — mieux vaut trop proposer que rien.
|
|
483
|
+
const classified = typed.chat.length + typed.embedding.length + typed.rerank.length;
|
|
484
|
+
if (classified === 0) {
|
|
485
|
+
return { ok: true, typed: false, source: 'models', chat: flat.models, embedding: flat.models, rerank: flat.models };
|
|
486
|
+
}
|
|
487
|
+
return {
|
|
488
|
+
ok: true,
|
|
489
|
+
typed: true,
|
|
490
|
+
source: 'models',
|
|
491
|
+
chat: typed.chat.length ? sortedUnique(typed.chat) : flat.models,
|
|
492
|
+
embedding: typed.embedding.length ? sortedUnique(typed.embedding) : flat.models,
|
|
493
|
+
rerank: typed.rerank.length ? sortedUnique(typed.rerank) : flat.models,
|
|
494
|
+
};
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
export function fallbackModels(engine, kind) {
|
|
498
|
+
return fallbackFor(engine, kind);
|
|
97
499
|
}
|