@doitian/dsh-provider-aliyun 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,431 @@
1
+ /**
2
+ * Ask the endpoint which models it serves.
3
+ *
4
+ * DSH's model picker reads a route's advertised models through
5
+ * `LlmAdapter.listModels()`, so a route answering from a file can only ever
6
+ * list what its package shipped. This module makes that answer live: one
7
+ * `GET {baseURL}/models`, the endpoint's own order, the endpoint's own ids.
8
+ *
9
+ * Three properties are deliberate, and each one is a failure mode avoided:
10
+ *
11
+ * - **No read waits longer than a bound.** `settle()` waits at most
12
+ * `maxWaitMs` for a cold or stale listing; every other read only *schedules*
13
+ * a refresh. A picker opened during a slow fetch shows the fallback catalog
14
+ * instead of hanging on the endpoint.
15
+ * - **A failure never shrinks the route.** The last good listing stays in
16
+ * effect, and before the first good one the fallback catalog does. Only a
17
+ * successful listing changes what the route advertises, and a listing naming
18
+ * nothing counts as a failure: `{"data":[]}` says nothing about what an
19
+ * endpoint serves, so acting on it would empty the picker for no reason.
20
+ * - **The credential is resolved per attempt.** The key can arrive long after
21
+ * mount — the credentials store is written at any time — so a failure is
22
+ * never cached as a conclusion; `credentials/reference-updated` and the
23
+ * failure backoff are what make the next attempt happen.
24
+ *
25
+ * @module @doitian/dsh-provider-aliyun/discovery
26
+ */
27
+ import { LlmError, assertUsableApiKey, attributionHeaders } from '@deepseek-ai/dsh-llm'
28
+
29
+ import { ROUTE } from './catalog.js'
30
+
31
+ /** Endpoint replies larger than this are refused outright; a model list is never this big. */
32
+ const MAX_RESPONSE_BYTES = 4 * 1024 * 1024
33
+
34
+ /** Idle bound on one listing request, so a black-holed endpoint cannot pin a read. */
35
+ export const DEFAULT_DISCOVERY_TIMEOUT_MS = 15_000
36
+
37
+ /** How long a failed attempt is left alone before another read may retry it. */
38
+ export const FAILURE_RETRY_MS = 60_000
39
+
40
+ /** Protocol this route speaks; the only listing shape this module reads. */
41
+ const PROTOCOL = 'openai-completions'
42
+
43
+ /**
44
+ * Join the configured endpoint with the listing path.
45
+ *
46
+ * The base is treated as a prefix rather than a URL to resolve against, so a
47
+ * workspace path keeps its segments instead of losing them to `URL` resolution:
48
+ * `https://…/compatible-mode/v1` lists at `https://…/compatible-mode/v1/models`.
49
+ *
50
+ * @param {string} baseURL - the route's configured endpoint.
51
+ * @returns {string} the listing URL.
52
+ */
53
+ export function listingUrl(baseURL) {
54
+ return `${baseURL.replace(/\/+$/, '')}/models`
55
+ }
56
+
57
+ /** The first non-empty string among the candidates. */
58
+ function label(...candidates) {
59
+ for (const candidate of candidates) {
60
+ if (typeof candidate === 'string' && candidate.length > 0) return candidate
61
+ }
62
+ return undefined
63
+ }
64
+
65
+ /** The first positive integer among the candidates. */
66
+ function capacity(...candidates) {
67
+ for (const candidate of candidates) {
68
+ if (typeof candidate === 'number' && Number.isInteger(candidate) && candidate > 0) return candidate
69
+ }
70
+ return undefined
71
+ }
72
+
73
+ /** Whether a value is a plain object, which is what a listing row must be. */
74
+ function isRow(value) {
75
+ return typeof value === 'object' && value !== null && !Array.isArray(value)
76
+ }
77
+
78
+ /**
79
+ * Read one model listing.
80
+ *
81
+ * Three shapes are accepted because compatible endpoints disagree: the standard
82
+ * `data` array (whose entries may be bare id strings), a `models` array, and the
83
+ * enriched `models` map some gateways expose, where the property key is the id
84
+ * the endpoint accepts and the nested `id` is a canonical name that may differ.
85
+ * A row without a usable id is skipped rather than failing the read — one
86
+ * malformed row should not deny the rest of a working endpoint.
87
+ *
88
+ * @param {unknown} body - the parsed reply.
89
+ * @returns {{ id: string, name: string, contextWindow?: number, maxTokens?: number }[]} rows in endpoint order, deduplicated.
90
+ * @throws {LlmError} when the reply is not a listing at all.
91
+ */
92
+ export function parseListing(body) {
93
+ const data = isRow(body) ? body.data : undefined
94
+ const models = isRow(body) ? body.models : undefined
95
+ let rows
96
+ if (Array.isArray(data)) rows = data
97
+ else if (Array.isArray(models)) rows = models
98
+ else if (isRow(models)) {
99
+ // The property key is the id the endpoint accepts; a nested `id` is only a
100
+ // canonical name for an entry whose key is empty, because gateways put the
101
+ // canonical identity there instead of the alias requests must use.
102
+ rows = Object.entries(models)
103
+ .filter(([, value]) => isRow(value))
104
+ .map(([key, value]) => ({ ...value, id: key.length > 0 ? key : value.id }))
105
+ } else {
106
+ throw new LlmError(
107
+ 'llm-aliyun: the endpoint\'s model listing has neither a "data" array nor a "models" object',
108
+ 'DISCOVERY_FAILED',
109
+ )
110
+ }
111
+
112
+ const seen = new Set()
113
+ const listing = []
114
+ for (const raw of rows) {
115
+ const row = typeof raw === 'string' ? { id: raw } : raw
116
+ if (!isRow(row)) continue
117
+ const id = label(row.id)
118
+ if (id === undefined || seen.has(id)) continue
119
+ seen.add(id)
120
+ const contextWindow = capacity(row.contextWindow, row.context_window, row.context_length, row.max_input_tokens, row.limit?.context)
121
+ const maxTokens = capacity(row.maxOutputTokens, row.max_output_tokens, row.maxTokens, row.max_tokens, row.limit?.output, row.top_provider?.max_completion_tokens)
122
+ listing.push({
123
+ id,
124
+ name: label(row.name, row.display_name, row.displayName) ?? id,
125
+ ...contextWindow === undefined ? {} : { contextWindow },
126
+ ...maxTokens === undefined ? {} : { maxTokens },
127
+ })
128
+ }
129
+ return listing
130
+ }
131
+
132
+ /**
133
+ * Read a reply body, refusing one that outgrows the ceiling.
134
+ *
135
+ * The declared length is checked first so an honest server is turned away
136
+ * without transferring anything, but the accumulated total is what actually
137
+ * enforces the bound: a server that under-declares tells us nothing up front.
138
+ *
139
+ * @param {Response} response - the listing reply.
140
+ * @param {string} url - the URL, for the message.
141
+ * @returns {Promise<string>} the decoded body.
142
+ */
143
+ async function readBounded(response, url) {
144
+ const oversized = () => new LlmError(`llm-aliyun: ${url} answered with more than ${MAX_RESPONSE_BYTES} bytes`, 'DISCOVERY_FAILED')
145
+ const declared = Number(response.headers.get('content-length') ?? Number.NaN)
146
+ if (Number.isFinite(declared) && declared > MAX_RESPONSE_BYTES) {
147
+ await response.body?.cancel().catch(() => {})
148
+ throw oversized()
149
+ }
150
+ if (response.body === null) return ''
151
+ const reader = response.body.getReader()
152
+ const chunks = []
153
+ let total = 0
154
+ try {
155
+ for (;;) {
156
+ const { done, value } = await reader.read()
157
+ if (done) break
158
+ total += value.byteLength
159
+ if (total > MAX_RESPONSE_BYTES) throw oversized()
160
+ chunks.push(value)
161
+ }
162
+ } finally {
163
+ await reader.cancel().catch(() => {})
164
+ }
165
+ const body = new Uint8Array(total)
166
+ let offset = 0
167
+ for (const chunk of chunks) {
168
+ body.set(chunk, offset)
169
+ offset += chunk.byteLength
170
+ }
171
+ return new TextDecoder().decode(body)
172
+ }
173
+
174
+ /**
175
+ * Fetch and parse one endpoint's model listing.
176
+ *
177
+ * @param {object} input - the request.
178
+ * @param {string} input.baseURL - endpoint to interrogate.
179
+ * @param {string} [input.apiKey] - credential for this request; omitted sends none.
180
+ * @param {Record<string, string>} [input.headers] - extra headers the deployment configures.
181
+ * @param {number} [input.timeoutMs] - idle bound on the request.
182
+ * @param {AbortSignal} [input.signal] - caller cancellation.
183
+ * @param {typeof fetch} [input.fetchImpl] - transport, injected by tests.
184
+ * @returns {Promise<{ id: string, name: string, contextWindow?: number, maxTokens?: number }[]>} the advertised models.
185
+ * @throws {LlmError} when the endpoint refuses, fails, or answers something that is not a listing.
186
+ */
187
+ export async function fetchListing({ baseURL, apiKey, headers, timeoutMs = DEFAULT_DISCOVERY_TIMEOUT_MS, signal, fetchImpl = globalThis.fetch }) {
188
+ const url = listingUrl(baseURL)
189
+ const requestHeaders = new Headers(headers === undefined ? undefined : Object.entries(headers))
190
+ requestHeaders.set('accept', 'application/json')
191
+ if (apiKey !== undefined) requestHeaders.set('authorization', `Bearer ${apiKey}`)
192
+ // Every provider HTTP request carries the harness attribution headers.
193
+ for (const [name, value] of Object.entries(attributionHeaders())) requestHeaders.set(name, value)
194
+
195
+ const bound = AbortSignal.timeout(timeoutMs)
196
+ const composed = signal === undefined ? bound : AbortSignal.any([signal, bound])
197
+ let response
198
+ try {
199
+ response = await fetchImpl(url, { method: 'GET', headers: requestHeaders, signal: composed })
200
+ } catch (error) {
201
+ const reached = signal?.aborted === true ? 'aborted by the caller' : `could not reach ${url}`
202
+ throw new LlmError(`llm-aliyun: ${reached}`, signal?.aborted === true ? 'ABORTED' : 'DISCOVERY_FAILED', { cause: error })
203
+ }
204
+ if (!response.ok) {
205
+ const advice = response.status === 401 || response.status === 403
206
+ ? '; check the stored credential'
207
+ : response.status === 404
208
+ ? '; this endpoint serves no model listing, so the fallback catalog is what this route advertises'
209
+ : ''
210
+ throw new LlmError(`llm-aliyun: ${url} answered ${response.status}${advice}`, 'DISCOVERY_FAILED')
211
+ }
212
+
213
+ const text = await readBounded(response, url)
214
+ let body
215
+ try {
216
+ body = JSON.parse(text)
217
+ } catch (error) {
218
+ throw new LlmError(`llm-aliyun: ${url} did not answer with JSON`, 'DISCOVERY_FAILED', { cause: error })
219
+ }
220
+ return parseListing(body)
221
+ }
222
+
223
+ /** Resolve after `ms`, without holding the process open for it. */
224
+ function delay(ms) {
225
+ return new Promise((resolve) => {
226
+ const timer = setTimeout(resolve, ms)
227
+ timer.unref?.()
228
+ })
229
+ }
230
+
231
+ /**
232
+ * Create the discovery state one mounted route owns.
233
+ *
234
+ * @param {object} options - route facts and callbacks.
235
+ * @param {string} options.baseURL - the route's configured endpoint.
236
+ * @param {string} options.apiKeyEnv - credential reference, named in diagnostics and validated.
237
+ * @param {boolean} [options.enabled] - whether reads refresh the listing at all.
238
+ * @param {number} [options.ttlMs] - how long a listing stays fresh.
239
+ * @param {number} [options.timeoutMs] - idle bound on one listing request.
240
+ * @param {Record<string, string>} [options.headers] - extra headers the deployment configures.
241
+ * @param {readonly import('./catalog.js').AliyunModel[]} [options.fallbackModels] - entries whose facts a probe may report.
242
+ * @param {() => Promise<string | undefined>} options.resolveApiKey - credential resolver.
243
+ * @param {(ids: readonly string[]) => void} [options.onChange] - called when the advertised membership changes.
244
+ * @param {(message: string) => void} [options.onError] - called once per distinct failure message.
245
+ * @param {typeof fetch} [options.fetchImpl] - transport, injected by tests.
246
+ * @param {() => number} [options.now] - clock, injected by tests.
247
+ * @returns {object} the discovery state.
248
+ */
249
+ export function createModelDiscovery(options) {
250
+ const {
251
+ baseURL,
252
+ apiKeyEnv,
253
+ enabled = true,
254
+ ttlMs,
255
+ timeoutMs = DEFAULT_DISCOVERY_TIMEOUT_MS,
256
+ headers,
257
+ fallbackModels = [],
258
+ resolveApiKey,
259
+ onChange,
260
+ onError,
261
+ fetchImpl,
262
+ now = () => Date.now(),
263
+ } = options
264
+
265
+ /** Endpoint order, `undefined` until a listing has ever been read. */
266
+ let ids
267
+ /** When the listing in effect was read. */
268
+ let fetchedAt = 0
269
+ /** When the last attempt started, successful or not, which is what paces retries. */
270
+ let attemptedAt = -Infinity
271
+ let failure
272
+ let reportedFailure
273
+ let inFlight
274
+
275
+ const report = (error) => {
276
+ failure = error instanceof Error ? error.message : String(error)
277
+ if (failure !== reportedFailure) {
278
+ reportedFailure = failure
279
+ onError?.(failure)
280
+ }
281
+ }
282
+
283
+ const succeed = (next) => {
284
+ failure = undefined
285
+ reportedFailure = undefined
286
+ fetchedAt = now()
287
+ const changed = ids === undefined || ids.length !== next.length || ids.some((id, index) => id !== next[index])
288
+ ids = next
289
+ if (changed) onChange?.(ids)
290
+ }
291
+
292
+ /** Whether another attempt is warranted right now. */
293
+ const due = () => {
294
+ if (ids === undefined) return now() - attemptedAt >= FAILURE_RETRY_MS
295
+ return now() - fetchedAt >= ttlMs
296
+ }
297
+
298
+ /**
299
+ * Read the listing unless one is fresh or in flight.
300
+ *
301
+ * Never rejects: a failure is reported through `onError` and leaves the
302
+ * listing in effect untouched, because a route that cannot reach its endpoint
303
+ * must still serve the models it already knows.
304
+ *
305
+ * @param {object} [options] - refresh options.
306
+ * @param {boolean} [options.force] - read even when the current listing is fresh.
307
+ * @returns {Promise<void>} settles when this attempt does.
308
+ */
309
+ const refresh = ({ force = false } = {}) => {
310
+ if (!enabled) return Promise.resolve()
311
+ if (inFlight !== undefined) return inFlight
312
+ if (!force && !due()) return Promise.resolve()
313
+ attemptedAt = now()
314
+ inFlight = (async () => {
315
+ try {
316
+ const raw = await resolveApiKey()
317
+ if (raw === undefined || raw.length === 0) {
318
+ throw new LlmError(
319
+ `llm-aliyun: no credential for provider route "${ROUTE}" — ${apiKeyEnv} is not set, so this route advertises its fallback catalog; store ${apiKeyEnv} to let it list what the endpoint serves`,
320
+ 'MISSING_CREDENTIAL',
321
+ )
322
+ }
323
+ const apiKey = assertUsableApiKey(raw, 'llm-aliyun', apiKeyEnv)
324
+ const listing = await fetchListing({ baseURL, apiKey, headers, timeoutMs, fetchImpl })
325
+ if (listing.length === 0) {
326
+ throw new LlmError(
327
+ `llm-aliyun: ${listingUrl(baseURL)} listed no models, which says nothing about what it serves; keeping the ${ids === undefined ? 'fallback catalog' : 'previous listing'}`,
328
+ 'DISCOVERY_FAILED',
329
+ )
330
+ }
331
+ succeed(listing.map((model) => model.id))
332
+ } catch (error) {
333
+ report(error)
334
+ } finally {
335
+ inFlight = undefined
336
+ }
337
+ })()
338
+ return inFlight
339
+ }
340
+
341
+ /** Schedule a refresh when one is due, without waiting for it. */
342
+ const ensureFresh = (refreshOptions) => {
343
+ void refresh(refreshOptions)
344
+ }
345
+
346
+ /**
347
+ * Answer a model-list read, waiting at most `maxWaitMs` for a listing that is
348
+ * not in hand yet.
349
+ *
350
+ * @param {object} [options] - wait options.
351
+ * @param {number} [options.maxWaitMs] - longest to wait; `0` only schedules.
352
+ * @returns {Promise<void>} settles when the wait is over, whether or not the read succeeded.
353
+ */
354
+ const settle = async ({ maxWaitMs = 0 } = {}) => {
355
+ if (!enabled || !due()) return
356
+ const pending = inFlight ?? refresh()
357
+ if (maxWaitMs <= 0) return
358
+ await Promise.race([pending, delay(maxWaitMs)])
359
+ }
360
+
361
+ /**
362
+ * Interrogate an endpoint on behalf of a configuration surface.
363
+ *
364
+ * This is what `ctx.llm.registerModelDiscovery` publishes: a draft carries its
365
+ * own endpoint and one-shot credential, while a request that names no endpoint
366
+ * is answered from this route's configuration. Capacities come from the
367
+ * fallback catalog when it names the id, so an adopting surface receives facts
368
+ * rather than having to invent them.
369
+ *
370
+ * @param {object} [request] - the draft being interrogated.
371
+ * @param {string} [request.provider] - route the draft edits, when it edits one.
372
+ * @param {string} [request.baseURL] - endpoint to interrogate.
373
+ * @param {string} [request.api] - protocol the draft names.
374
+ * @param {string} [request.apiKey] - credential for this interrogation alone.
375
+ * @param {AbortSignal} [signal] - caller cancellation.
376
+ * @returns {Promise<object[]>} the advertised models, in endpoint order.
377
+ * @throws {LlmError} when the request cannot be served or the endpoint refuses.
378
+ */
379
+ const probe = async (request = {}, signal) => {
380
+ const provider = label(request.provider)
381
+ const endpoint = label(request.baseURL)
382
+ const api = label(request.api)
383
+ if (api !== undefined && api !== PROTOCOL) {
384
+ throw new LlmError(`llm-aliyun: this route speaks ${PROTOCOL}, not "${api}", and cannot list models for it`, 'DISCOVERY_UNSUPPORTED')
385
+ }
386
+ if (endpoint === undefined && provider !== undefined && provider !== 'aliyun') {
387
+ throw new LlmError(`llm-aliyun: the draft names route "${provider}", which this plugin does not own; give its endpoint to interrogate it`, 'DISCOVERY_UNSUPPORTED')
388
+ }
389
+ const supplied = label(request.apiKey)
390
+ const apiKey = supplied === undefined ? await resolveApiKey() : supplied
391
+ const listing = await fetchListing({
392
+ baseURL: endpoint ?? baseURL,
393
+ ...apiKey === undefined || apiKey.length === 0 ? {} : { apiKey: assertUsableApiKey(apiKey, 'llm-aliyun', apiKeyEnv) },
394
+ headers,
395
+ timeoutMs,
396
+ signal,
397
+ fetchImpl,
398
+ })
399
+ const known = new Map(fallbackModels.map((entry) => [entry.id, entry]))
400
+ return listing.map((model) => {
401
+ const entry = known.get(model.id)
402
+ const contextWindow = model.contextWindow ?? entry?.contextWindow
403
+ const maxTokens = model.maxTokens ?? entry?.maxTokens
404
+ const input = entry?.input
405
+ return {
406
+ id: model.id,
407
+ name: model.name,
408
+ ...contextWindow === undefined ? {} : { contextWindow },
409
+ ...maxTokens === undefined ? {} : { maxTokens },
410
+ ...input === undefined ? {} : { inputModalities: [...input] },
411
+ }
412
+ })
413
+ }
414
+
415
+ return {
416
+ /** Whether reads refresh at all. */
417
+ get enabled() {
418
+ return enabled
419
+ },
420
+ /** The ids in effect, or `undefined` before any listing. */
421
+ ids: () => ids,
422
+ /** When the listing in effect was read, or `0`. */
423
+ fetchedAt: () => fetchedAt,
424
+ /** The last failure message, cleared by a successful read. */
425
+ error: () => failure,
426
+ refresh,
427
+ ensureFresh,
428
+ settle,
429
+ probe,
430
+ }
431
+ }
package/lib/index.js CHANGED
@@ -2,9 +2,15 @@
2
2
  * Aliyun DashScope (Bailian) as a DeepSeek Harness model provider.
3
3
  *
4
4
  * This plugin registers one provider route, `aliyun`, on the harness LLM seam.
5
- * It owns the route outright — endpoint, credential reference, protocol, and
6
- * the Qwen model catalog in {@link ./catalog.js} — so a profile needs no model
7
- * list, and `pnpm update` is how the catalog moves forward.
5
+ * It owns the route outright — endpoint, credential reference, protocol, and the
6
+ * model catalog — so a profile needs no model list, and `pnpm update` is how the
7
+ * package moves forward.
8
+ *
9
+ * The catalog it advertises is live rather than shipped: {@link ./discovery.js}
10
+ * asks the configured endpoint which models it serves, {@link ./catalog.js}
11
+ * supplies the fallback membership and the model facts, and {@link ./models.js}
12
+ * decides what those two add up to. A profile can still pin the whole thing by
13
+ * setting `models`, and `discovery.enabled: false` turns the live half off.
8
14
  *
9
15
  * @module @doitian/dsh-provider-aliyun
10
16
  */
@@ -13,12 +19,22 @@ import z from '@deepseek-ai/schemastery'
13
19
 
14
20
  import { createAliyunAdapter } from './adapter.js'
15
21
  import {
16
- ALIYUN_MODELS,
22
+ CHAT_MODEL_PATTERNS,
17
23
  DEFAULT_API_KEY_ENV,
18
24
  DEFAULT_BASE_URL,
25
+ DISCOVERY_DEFAULTS,
26
+ DISCOVERY_TTL_MS,
27
+ DISCOVERY_WAIT_MS,
19
28
  DISPLAY_NAME,
29
+ FALLBACK_MODELS,
30
+ NON_CHAT_MODEL_PATTERNS,
20
31
  ROUTE,
21
32
  } from './catalog.js'
33
+ import { DEFAULT_DISCOVERY_TIMEOUT_MS, createModelDiscovery } from './discovery.js'
34
+ // A build-time snapshot of Alibaba model facts; `npm run generate:metadata`
35
+ // refreshes it. Nothing at run time talks to the registry it came from.
36
+ import metadataSnapshot from './metadata.json' with { type: 'json' }
37
+ import { compilePatterns, createMetadataIndex, mergeCatalog, selectIds } from './models.js'
22
38
 
23
39
  /** Entry name, used for logging and as the settings namespace fallback. */
24
40
  export const name = 'llm-aliyun'
@@ -48,16 +64,29 @@ const modelSchema = z.object({
48
64
  *
49
65
  * Every field defaults to something serviceable, so the bundle's patch mounts
50
66
  * this plugin with an empty config and the route works as soon as a key
51
- * resolves. `models` defaults to the shipped catalog, which is what the
52
- * settings surface shows as inherited rows until someone edits them.
67
+ * resolves. `models` is the fallback catalog *and* the metadata table: discovery
68
+ * decides which ids the route advertises, and an id named here keeps the
69
+ * capacities written here.
53
70
  */
54
71
  export const Config = z.object({
55
72
  displayName: z.string().default(DISPLAY_NAME),
56
73
  apiKeyEnv: z.string().role('credential-ref').default(DEFAULT_API_KEY_ENV),
57
74
  baseURL: z.string().default(DEFAULT_BASE_URL),
58
- models: z.array(modelSchema).default(ALIYUN_MODELS),
75
+ models: z.array(modelSchema).default(FALLBACK_MODELS),
59
76
  reasoning: z.union(THINKING_LEVELS),
60
77
  headers: z.dict(z.string()),
78
+ discovery: z.object({
79
+ enabled: z.boolean().default(true),
80
+ filter: z.union(['chat', 'patterns', 'all']).default('chat'),
81
+ include: z.array(z.string()).default([...CHAT_MODEL_PATTERNS]),
82
+ exclude: z.array(z.string()).default([...NON_CHAT_MODEL_PATTERNS]),
83
+ ttlMs: z.number().step(1).min(0).default(DISCOVERY_TTL_MS),
84
+ waitMs: z.number().step(1).min(0).default(DISCOVERY_WAIT_MS),
85
+ timeoutMs: z.number().step(1).min(1).default(DEFAULT_DISCOVERY_TIMEOUT_MS),
86
+ contextWindow: z.number().step(1).min(1).default(DISCOVERY_DEFAULTS.contextWindow),
87
+ maxTokens: z.number().step(1).min(1).default(DISCOVERY_DEFAULTS.maxTokens),
88
+ input: z.array(z.union(MODALITIES)).default([...DISCOVERY_DEFAULTS.input]),
89
+ }),
61
90
  })
62
91
 
63
92
  /**
@@ -96,15 +125,40 @@ export function apply(ctx, config) {
96
125
  // The settings namespace is the entry id, so the Models page addresses this
97
126
  // plugin's own row rather than any other adapter's section.
98
127
  const settingsNs = ctx.fiber?.entry?.options.id ?? name
128
+ const discoveryConfig = config.discovery
129
+ const fallback = config.models
130
+ const resolveApiKey = apiKeyResolver(ctx)
131
+
132
+ // A workspace listing is the account's whole catalogue, so what it is allowed
133
+ // to advertise is filtered before it becomes a catalog. Two signals decide
134
+ // that — the profile's patterns and the metadata snapshot's own statement that
135
+ // a model answers with text — and a pattern that does not compile costs itself
136
+ // and nothing else.
137
+ const metadata = createMetadataIndex(metadataSnapshot)
138
+ const complain = (field) => (source, error) => {
139
+ ctx.logger?.warn?.(`llm-aliyun: discovery.${field} pattern ${JSON.stringify(source)} is not a valid regular expression (${error.message}); ignoring it`)
140
+ }
141
+ const filter = {
142
+ mode: discoveryConfig.filter,
143
+ include: compilePatterns(discoveryConfig.include, complain('include')),
144
+ exclude: compilePatterns(discoveryConfig.exclude, complain('exclude')),
145
+ // Naming a model in `models` is a stronger statement than any pattern.
146
+ pinned: fallback.map((entry) => entry.id),
147
+ known: (id) => metadata.knows(id),
148
+ }
149
+ ctx.logger?.debug?.(`llm-aliyun: model metadata snapshot from ${metadataSnapshot.source} (${metadata.size} models, generated ${metadataSnapshot.generatedAt})`)
99
150
 
100
- const adapter = createAliyunAdapter({
151
+ // The adapter is built first only because discovery's change callback needs
152
+ // `setCatalog`; its own discovery callbacks are closures that run later.
153
+ const { adapter, setCatalog } = createAliyunAdapter({
101
154
  displayName: config.displayName,
102
155
  apiKeyEnv: config.apiKeyEnv,
103
156
  baseURL: config.baseURL,
104
- models: config.models,
157
+ // What the route advertises until a listing arrives.
158
+ models: fallback,
105
159
  reasoning: config.reasoning,
106
160
  headers: config.headers,
107
- resolveApiKey: apiKeyResolver(ctx),
161
+ resolveApiKey,
108
162
  resolveAttachments: () => ctx.get('attachments'),
109
163
  resolveImageAccess: (attachments, ref) => resolveImageAttachmentAccess(
110
164
  attachments,
@@ -114,15 +168,55 @@ export function apply(ctx, config) {
114
168
  onReplayDegrade: ({ provider, model, reason }) => {
115
169
  ctx.logger?.warn?.(`llm-aliyun: unusable replay state on assistant history for route "${provider}/${model}"; sending that message as provider-neutral content (${reason})`)
116
170
  },
171
+ // A read is the moment a stale listing is noticed, and a model list is the
172
+ // one read allowed to wait for the endpoint's answer.
173
+ onRead: () => discovery.ensureFresh(),
174
+ settle: () => discovery.settle({ maxWaitMs: discoveryConfig.waitMs }),
175
+ })
176
+
177
+ /**
178
+ * Advertise a new listing, and tell the surfaces that read the old one.
179
+ *
180
+ * A model picker caches the catalog it read and re-reads it on
181
+ * `llm/adapters-updated` — the payload-free event the LLM seam publishes when
182
+ * a route set changes. Swapping the catalog alone is invisible to it, so the
183
+ * picker would keep showing whatever it read at startup until something else
184
+ * moved. Re-registering the same route is the documented way to publish that
185
+ * event: it changes no route, only the topology version the event announces.
186
+ */
187
+ const announceCatalog = (models) => {
188
+ setCatalog(models)
189
+ registration.replace([ROUTE])
190
+ }
191
+
192
+ const discovery = createModelDiscovery({
193
+ baseURL: config.baseURL,
194
+ apiKeyEnv: config.apiKeyEnv,
195
+ enabled: discoveryConfig.enabled,
196
+ ttlMs: discoveryConfig.ttlMs,
197
+ timeoutMs: discoveryConfig.timeoutMs,
198
+ headers: config.headers,
199
+ fallbackModels: fallback,
200
+ resolveApiKey: () => resolveApiKey(ROUTE, { apiKeyEnv: config.apiKeyEnv }),
201
+ onChange: (ids) => announceCatalog(mergeCatalog({
202
+ ids: selectIds(ids, filter),
203
+ fallback,
204
+ defaults: {
205
+ contextWindow: discoveryConfig.contextWindow,
206
+ maxTokens: discoveryConfig.maxTokens,
207
+ input: discoveryConfig.input,
208
+ },
209
+ metadata,
210
+ })),
211
+ onError: (message) => ctx.logger?.warn?.(message),
117
212
  })
118
213
 
119
214
  // Both registrations are disposed with the plugin's fiber.
120
- ctx.llm.registerAdapter([ROUTE], adapter)
215
+ const registration = ctx.llm.registerAdapter([ROUTE], adapter)
121
216
 
122
- // The directory entry is what gives the route a row on the Models page, and
123
- // with it the API-key field. It is refused if another adapter already
124
- // declares `aliyun` — which is what happens when a profile still configures
125
- // an `aliyun` route through `llm-pi-ai`; remove that route to use this plugin.
217
+ // The directory entry is what gives the route a row on the Models page. It is
218
+ // refused if another adapter already declares `aliyun` — which is what happens
219
+ // when a profile still configures an `aliyun` route through `llm-pi-ai`.
126
220
  ctx.llm.registerConfigurableProviders([{
127
221
  provider: ROUTE,
128
222
  displayName: config.displayName,
@@ -130,4 +224,16 @@ export function apply(ctx, config) {
130
224
  settingsPath: [],
131
225
  declared: true,
132
226
  }])
227
+
228
+ // A configuration surface can interrogate this route's endpoint — even before
229
+ // a key is stored, by passing the draft's own.
230
+ ctx.llm.registerModelDiscovery(settingsNs, (request, signal) => discovery.probe(request, signal))
231
+
232
+ // The key can arrive at any time, and a stored key is exactly the event that
233
+ // makes a failed listing attempt worth retrying immediately.
234
+ ctx.on('credentials/reference-updated', (ref) => {
235
+ if (typeof ref === 'string' && ref === config.apiKeyEnv) discovery.ensureFresh({ force: true })
236
+ })
237
+
238
+ discovery.ensureFresh()
133
239
  }