@doitian/dsh-provider-aliyun 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +126 -27
- package/lib/adapter.js +49 -6
- package/lib/catalog.js +110 -41
- package/lib/discovery.js +431 -0
- package/lib/index.js +121 -15
- package/lib/metadata.json +927 -0
- package/lib/models.js +168 -2
- package/package.json +8 -2
package/lib/discovery.js
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Ask the endpoint which models it serves.
|
|
3
|
+
*
|
|
4
|
+
* DSH's model picker reads a route's advertised models through
|
|
5
|
+
* `LlmAdapter.listModels()`, so a route answering from a file can only ever
|
|
6
|
+
* list what its package shipped. This module makes that answer live: one
|
|
7
|
+
* `GET {baseURL}/models`, the endpoint's own order, the endpoint's own ids.
|
|
8
|
+
*
|
|
9
|
+
* Three properties are deliberate, and each one is a failure mode avoided:
|
|
10
|
+
*
|
|
11
|
+
* - **No read waits longer than a bound.** `settle()` waits at most
|
|
12
|
+
* `maxWaitMs` for a cold or stale listing; every other read only *schedules*
|
|
13
|
+
* a refresh. A picker opened during a slow fetch shows the fallback catalog
|
|
14
|
+
* instead of hanging on the endpoint.
|
|
15
|
+
* - **A failure never shrinks the route.** The last good listing stays in
|
|
16
|
+
* effect, and before the first good one the fallback catalog does. Only a
|
|
17
|
+
* successful listing changes what the route advertises, and a listing naming
|
|
18
|
+
* nothing counts as a failure: `{"data":[]}` says nothing about what an
|
|
19
|
+
* endpoint serves, so acting on it would empty the picker for no reason.
|
|
20
|
+
* - **The credential is resolved per attempt.** The key can arrive long after
|
|
21
|
+
* mount — the credentials store is written at any time — so a failure is
|
|
22
|
+
* never cached as a conclusion; `credentials/reference-updated` and the
|
|
23
|
+
* failure backoff are what make the next attempt happen.
|
|
24
|
+
*
|
|
25
|
+
* @module @doitian/dsh-provider-aliyun/discovery
|
|
26
|
+
*/
|
|
27
|
+
import { LlmError, assertUsableApiKey, attributionHeaders } from '@deepseek-ai/dsh-llm'
|
|
28
|
+
|
|
29
|
+
import { ROUTE } from './catalog.js'
|
|
30
|
+
|
|
31
|
+
/** Endpoint replies larger than this are refused outright; a model list is never this big. */
|
|
32
|
+
const MAX_RESPONSE_BYTES = 4 * 1024 * 1024
|
|
33
|
+
|
|
34
|
+
/** Idle bound on one listing request, so a black-holed endpoint cannot pin a read. */
|
|
35
|
+
export const DEFAULT_DISCOVERY_TIMEOUT_MS = 15_000
|
|
36
|
+
|
|
37
|
+
/** How long a failed attempt is left alone before another read may retry it. */
|
|
38
|
+
export const FAILURE_RETRY_MS = 60_000
|
|
39
|
+
|
|
40
|
+
/** Protocol this route speaks; the only listing shape this module reads. */
|
|
41
|
+
const PROTOCOL = 'openai-completions'
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Join the configured endpoint with the listing path.
|
|
45
|
+
*
|
|
46
|
+
* The base is treated as a prefix rather than a URL to resolve against, so a
|
|
47
|
+
* workspace path keeps its segments instead of losing them to `URL` resolution:
|
|
48
|
+
* `https://…/compatible-mode/v1` lists at `https://…/compatible-mode/v1/models`.
|
|
49
|
+
*
|
|
50
|
+
* @param {string} baseURL - the route's configured endpoint.
|
|
51
|
+
* @returns {string} the listing URL.
|
|
52
|
+
*/
|
|
53
|
+
export function listingUrl(baseURL) {
|
|
54
|
+
return `${baseURL.replace(/\/+$/, '')}/models`
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/** The first non-empty string among the candidates. */
|
|
58
|
+
function label(...candidates) {
|
|
59
|
+
for (const candidate of candidates) {
|
|
60
|
+
if (typeof candidate === 'string' && candidate.length > 0) return candidate
|
|
61
|
+
}
|
|
62
|
+
return undefined
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** The first positive integer among the candidates. */
|
|
66
|
+
function capacity(...candidates) {
|
|
67
|
+
for (const candidate of candidates) {
|
|
68
|
+
if (typeof candidate === 'number' && Number.isInteger(candidate) && candidate > 0) return candidate
|
|
69
|
+
}
|
|
70
|
+
return undefined
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** Whether a value is a plain object, which is what a listing row must be. */
|
|
74
|
+
function isRow(value) {
|
|
75
|
+
return typeof value === 'object' && value !== null && !Array.isArray(value)
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Read one model listing.
|
|
80
|
+
*
|
|
81
|
+
* Three shapes are accepted because compatible endpoints disagree: the standard
|
|
82
|
+
* `data` array (whose entries may be bare id strings), a `models` array, and the
|
|
83
|
+
* enriched `models` map some gateways expose, where the property key is the id
|
|
84
|
+
* the endpoint accepts and the nested `id` is a canonical name that may differ.
|
|
85
|
+
* A row without a usable id is skipped rather than failing the read — one
|
|
86
|
+
* malformed row should not deny the rest of a working endpoint.
|
|
87
|
+
*
|
|
88
|
+
* @param {unknown} body - the parsed reply.
|
|
89
|
+
* @returns {{ id: string, name: string, contextWindow?: number, maxTokens?: number }[]} rows in endpoint order, deduplicated.
|
|
90
|
+
* @throws {LlmError} when the reply is not a listing at all.
|
|
91
|
+
*/
|
|
92
|
+
export function parseListing(body) {
|
|
93
|
+
const data = isRow(body) ? body.data : undefined
|
|
94
|
+
const models = isRow(body) ? body.models : undefined
|
|
95
|
+
let rows
|
|
96
|
+
if (Array.isArray(data)) rows = data
|
|
97
|
+
else if (Array.isArray(models)) rows = models
|
|
98
|
+
else if (isRow(models)) {
|
|
99
|
+
// The property key is the id the endpoint accepts; a nested `id` is only a
|
|
100
|
+
// canonical name for an entry whose key is empty, because gateways put the
|
|
101
|
+
// canonical identity there instead of the alias requests must use.
|
|
102
|
+
rows = Object.entries(models)
|
|
103
|
+
.filter(([, value]) => isRow(value))
|
|
104
|
+
.map(([key, value]) => ({ ...value, id: key.length > 0 ? key : value.id }))
|
|
105
|
+
} else {
|
|
106
|
+
throw new LlmError(
|
|
107
|
+
'llm-aliyun: the endpoint\'s model listing has neither a "data" array nor a "models" object',
|
|
108
|
+
'DISCOVERY_FAILED',
|
|
109
|
+
)
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
const seen = new Set()
|
|
113
|
+
const listing = []
|
|
114
|
+
for (const raw of rows) {
|
|
115
|
+
const row = typeof raw === 'string' ? { id: raw } : raw
|
|
116
|
+
if (!isRow(row)) continue
|
|
117
|
+
const id = label(row.id)
|
|
118
|
+
if (id === undefined || seen.has(id)) continue
|
|
119
|
+
seen.add(id)
|
|
120
|
+
const contextWindow = capacity(row.contextWindow, row.context_window, row.context_length, row.max_input_tokens, row.limit?.context)
|
|
121
|
+
const maxTokens = capacity(row.maxOutputTokens, row.max_output_tokens, row.maxTokens, row.max_tokens, row.limit?.output, row.top_provider?.max_completion_tokens)
|
|
122
|
+
listing.push({
|
|
123
|
+
id,
|
|
124
|
+
name: label(row.name, row.display_name, row.displayName) ?? id,
|
|
125
|
+
...contextWindow === undefined ? {} : { contextWindow },
|
|
126
|
+
...maxTokens === undefined ? {} : { maxTokens },
|
|
127
|
+
})
|
|
128
|
+
}
|
|
129
|
+
return listing
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Read a reply body, refusing one that outgrows the ceiling.
|
|
134
|
+
*
|
|
135
|
+
* The declared length is checked first so an honest server is turned away
|
|
136
|
+
* without transferring anything, but the accumulated total is what actually
|
|
137
|
+
* enforces the bound: a server that under-declares tells us nothing up front.
|
|
138
|
+
*
|
|
139
|
+
* @param {Response} response - the listing reply.
|
|
140
|
+
* @param {string} url - the URL, for the message.
|
|
141
|
+
* @returns {Promise<string>} the decoded body.
|
|
142
|
+
*/
|
|
143
|
+
async function readBounded(response, url) {
|
|
144
|
+
const oversized = () => new LlmError(`llm-aliyun: ${url} answered with more than ${MAX_RESPONSE_BYTES} bytes`, 'DISCOVERY_FAILED')
|
|
145
|
+
const declared = Number(response.headers.get('content-length') ?? Number.NaN)
|
|
146
|
+
if (Number.isFinite(declared) && declared > MAX_RESPONSE_BYTES) {
|
|
147
|
+
await response.body?.cancel().catch(() => {})
|
|
148
|
+
throw oversized()
|
|
149
|
+
}
|
|
150
|
+
if (response.body === null) return ''
|
|
151
|
+
const reader = response.body.getReader()
|
|
152
|
+
const chunks = []
|
|
153
|
+
let total = 0
|
|
154
|
+
try {
|
|
155
|
+
for (;;) {
|
|
156
|
+
const { done, value } = await reader.read()
|
|
157
|
+
if (done) break
|
|
158
|
+
total += value.byteLength
|
|
159
|
+
if (total > MAX_RESPONSE_BYTES) throw oversized()
|
|
160
|
+
chunks.push(value)
|
|
161
|
+
}
|
|
162
|
+
} finally {
|
|
163
|
+
await reader.cancel().catch(() => {})
|
|
164
|
+
}
|
|
165
|
+
const body = new Uint8Array(total)
|
|
166
|
+
let offset = 0
|
|
167
|
+
for (const chunk of chunks) {
|
|
168
|
+
body.set(chunk, offset)
|
|
169
|
+
offset += chunk.byteLength
|
|
170
|
+
}
|
|
171
|
+
return new TextDecoder().decode(body)
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Fetch and parse one endpoint's model listing.
|
|
176
|
+
*
|
|
177
|
+
* @param {object} input - the request.
|
|
178
|
+
* @param {string} input.baseURL - endpoint to interrogate.
|
|
179
|
+
* @param {string} [input.apiKey] - credential for this request; omitted sends none.
|
|
180
|
+
* @param {Record<string, string>} [input.headers] - extra headers the deployment configures.
|
|
181
|
+
* @param {number} [input.timeoutMs] - idle bound on the request.
|
|
182
|
+
* @param {AbortSignal} [input.signal] - caller cancellation.
|
|
183
|
+
* @param {typeof fetch} [input.fetchImpl] - transport, injected by tests.
|
|
184
|
+
* @returns {Promise<{ id: string, name: string, contextWindow?: number, maxTokens?: number }[]>} the advertised models.
|
|
185
|
+
* @throws {LlmError} when the endpoint refuses, fails, or answers something that is not a listing.
|
|
186
|
+
*/
|
|
187
|
+
export async function fetchListing({ baseURL, apiKey, headers, timeoutMs = DEFAULT_DISCOVERY_TIMEOUT_MS, signal, fetchImpl = globalThis.fetch }) {
|
|
188
|
+
const url = listingUrl(baseURL)
|
|
189
|
+
const requestHeaders = new Headers(headers === undefined ? undefined : Object.entries(headers))
|
|
190
|
+
requestHeaders.set('accept', 'application/json')
|
|
191
|
+
if (apiKey !== undefined) requestHeaders.set('authorization', `Bearer ${apiKey}`)
|
|
192
|
+
// Every provider HTTP request carries the harness attribution headers.
|
|
193
|
+
for (const [name, value] of Object.entries(attributionHeaders())) requestHeaders.set(name, value)
|
|
194
|
+
|
|
195
|
+
const bound = AbortSignal.timeout(timeoutMs)
|
|
196
|
+
const composed = signal === undefined ? bound : AbortSignal.any([signal, bound])
|
|
197
|
+
let response
|
|
198
|
+
try {
|
|
199
|
+
response = await fetchImpl(url, { method: 'GET', headers: requestHeaders, signal: composed })
|
|
200
|
+
} catch (error) {
|
|
201
|
+
const reached = signal?.aborted === true ? 'aborted by the caller' : `could not reach ${url}`
|
|
202
|
+
throw new LlmError(`llm-aliyun: ${reached}`, signal?.aborted === true ? 'ABORTED' : 'DISCOVERY_FAILED', { cause: error })
|
|
203
|
+
}
|
|
204
|
+
if (!response.ok) {
|
|
205
|
+
const advice = response.status === 401 || response.status === 403
|
|
206
|
+
? '; check the stored credential'
|
|
207
|
+
: response.status === 404
|
|
208
|
+
? '; this endpoint serves no model listing, so the fallback catalog is what this route advertises'
|
|
209
|
+
: ''
|
|
210
|
+
throw new LlmError(`llm-aliyun: ${url} answered ${response.status}${advice}`, 'DISCOVERY_FAILED')
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
const text = await readBounded(response, url)
|
|
214
|
+
let body
|
|
215
|
+
try {
|
|
216
|
+
body = JSON.parse(text)
|
|
217
|
+
} catch (error) {
|
|
218
|
+
throw new LlmError(`llm-aliyun: ${url} did not answer with JSON`, 'DISCOVERY_FAILED', { cause: error })
|
|
219
|
+
}
|
|
220
|
+
return parseListing(body)
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/** Resolve after `ms`, without holding the process open for it. */
|
|
224
|
+
function delay(ms) {
|
|
225
|
+
return new Promise((resolve) => {
|
|
226
|
+
const timer = setTimeout(resolve, ms)
|
|
227
|
+
timer.unref?.()
|
|
228
|
+
})
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* Create the discovery state one mounted route owns.
|
|
233
|
+
*
|
|
234
|
+
* @param {object} options - route facts and callbacks.
|
|
235
|
+
* @param {string} options.baseURL - the route's configured endpoint.
|
|
236
|
+
* @param {string} options.apiKeyEnv - credential reference, named in diagnostics and validated.
|
|
237
|
+
* @param {boolean} [options.enabled] - whether reads refresh the listing at all.
|
|
238
|
+
* @param {number} [options.ttlMs] - how long a listing stays fresh.
|
|
239
|
+
* @param {number} [options.timeoutMs] - idle bound on one listing request.
|
|
240
|
+
* @param {Record<string, string>} [options.headers] - extra headers the deployment configures.
|
|
241
|
+
* @param {readonly import('./catalog.js').AliyunModel[]} [options.fallbackModels] - entries whose facts a probe may report.
|
|
242
|
+
* @param {() => Promise<string | undefined>} options.resolveApiKey - credential resolver.
|
|
243
|
+
* @param {(ids: readonly string[]) => void} [options.onChange] - called when the advertised membership changes.
|
|
244
|
+
* @param {(message: string) => void} [options.onError] - called once per distinct failure message.
|
|
245
|
+
* @param {typeof fetch} [options.fetchImpl] - transport, injected by tests.
|
|
246
|
+
* @param {() => number} [options.now] - clock, injected by tests.
|
|
247
|
+
* @returns {object} the discovery state.
|
|
248
|
+
*/
|
|
249
|
+
export function createModelDiscovery(options) {
|
|
250
|
+
const {
|
|
251
|
+
baseURL,
|
|
252
|
+
apiKeyEnv,
|
|
253
|
+
enabled = true,
|
|
254
|
+
ttlMs,
|
|
255
|
+
timeoutMs = DEFAULT_DISCOVERY_TIMEOUT_MS,
|
|
256
|
+
headers,
|
|
257
|
+
fallbackModels = [],
|
|
258
|
+
resolveApiKey,
|
|
259
|
+
onChange,
|
|
260
|
+
onError,
|
|
261
|
+
fetchImpl,
|
|
262
|
+
now = () => Date.now(),
|
|
263
|
+
} = options
|
|
264
|
+
|
|
265
|
+
/** Endpoint order, `undefined` until a listing has ever been read. */
|
|
266
|
+
let ids
|
|
267
|
+
/** When the listing in effect was read. */
|
|
268
|
+
let fetchedAt = 0
|
|
269
|
+
/** When the last attempt started, successful or not, which is what paces retries. */
|
|
270
|
+
let attemptedAt = -Infinity
|
|
271
|
+
let failure
|
|
272
|
+
let reportedFailure
|
|
273
|
+
let inFlight
|
|
274
|
+
|
|
275
|
+
const report = (error) => {
|
|
276
|
+
failure = error instanceof Error ? error.message : String(error)
|
|
277
|
+
if (failure !== reportedFailure) {
|
|
278
|
+
reportedFailure = failure
|
|
279
|
+
onError?.(failure)
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
const succeed = (next) => {
|
|
284
|
+
failure = undefined
|
|
285
|
+
reportedFailure = undefined
|
|
286
|
+
fetchedAt = now()
|
|
287
|
+
const changed = ids === undefined || ids.length !== next.length || ids.some((id, index) => id !== next[index])
|
|
288
|
+
ids = next
|
|
289
|
+
if (changed) onChange?.(ids)
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
/** Whether another attempt is warranted right now. */
|
|
293
|
+
const due = () => {
|
|
294
|
+
if (ids === undefined) return now() - attemptedAt >= FAILURE_RETRY_MS
|
|
295
|
+
return now() - fetchedAt >= ttlMs
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
/**
|
|
299
|
+
* Read the listing unless one is fresh or in flight.
|
|
300
|
+
*
|
|
301
|
+
* Never rejects: a failure is reported through `onError` and leaves the
|
|
302
|
+
* listing in effect untouched, because a route that cannot reach its endpoint
|
|
303
|
+
* must still serve the models it already knows.
|
|
304
|
+
*
|
|
305
|
+
* @param {object} [options] - refresh options.
|
|
306
|
+
* @param {boolean} [options.force] - read even when the current listing is fresh.
|
|
307
|
+
* @returns {Promise<void>} settles when this attempt does.
|
|
308
|
+
*/
|
|
309
|
+
const refresh = ({ force = false } = {}) => {
|
|
310
|
+
if (!enabled) return Promise.resolve()
|
|
311
|
+
if (inFlight !== undefined) return inFlight
|
|
312
|
+
if (!force && !due()) return Promise.resolve()
|
|
313
|
+
attemptedAt = now()
|
|
314
|
+
inFlight = (async () => {
|
|
315
|
+
try {
|
|
316
|
+
const raw = await resolveApiKey()
|
|
317
|
+
if (raw === undefined || raw.length === 0) {
|
|
318
|
+
throw new LlmError(
|
|
319
|
+
`llm-aliyun: no credential for provider route "${ROUTE}" — ${apiKeyEnv} is not set, so this route advertises its fallback catalog; store ${apiKeyEnv} to let it list what the endpoint serves`,
|
|
320
|
+
'MISSING_CREDENTIAL',
|
|
321
|
+
)
|
|
322
|
+
}
|
|
323
|
+
const apiKey = assertUsableApiKey(raw, 'llm-aliyun', apiKeyEnv)
|
|
324
|
+
const listing = await fetchListing({ baseURL, apiKey, headers, timeoutMs, fetchImpl })
|
|
325
|
+
if (listing.length === 0) {
|
|
326
|
+
throw new LlmError(
|
|
327
|
+
`llm-aliyun: ${listingUrl(baseURL)} listed no models, which says nothing about what it serves; keeping the ${ids === undefined ? 'fallback catalog' : 'previous listing'}`,
|
|
328
|
+
'DISCOVERY_FAILED',
|
|
329
|
+
)
|
|
330
|
+
}
|
|
331
|
+
succeed(listing.map((model) => model.id))
|
|
332
|
+
} catch (error) {
|
|
333
|
+
report(error)
|
|
334
|
+
} finally {
|
|
335
|
+
inFlight = undefined
|
|
336
|
+
}
|
|
337
|
+
})()
|
|
338
|
+
return inFlight
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
/** Schedule a refresh when one is due, without waiting for it. */
|
|
342
|
+
const ensureFresh = (refreshOptions) => {
|
|
343
|
+
void refresh(refreshOptions)
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
/**
|
|
347
|
+
* Answer a model-list read, waiting at most `maxWaitMs` for a listing that is
|
|
348
|
+
* not in hand yet.
|
|
349
|
+
*
|
|
350
|
+
* @param {object} [options] - wait options.
|
|
351
|
+
* @param {number} [options.maxWaitMs] - longest to wait; `0` only schedules.
|
|
352
|
+
* @returns {Promise<void>} settles when the wait is over, whether or not the read succeeded.
|
|
353
|
+
*/
|
|
354
|
+
const settle = async ({ maxWaitMs = 0 } = {}) => {
|
|
355
|
+
if (!enabled || !due()) return
|
|
356
|
+
const pending = inFlight ?? refresh()
|
|
357
|
+
if (maxWaitMs <= 0) return
|
|
358
|
+
await Promise.race([pending, delay(maxWaitMs)])
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
/**
|
|
362
|
+
* Interrogate an endpoint on behalf of a configuration surface.
|
|
363
|
+
*
|
|
364
|
+
* This is what `ctx.llm.registerModelDiscovery` publishes: a draft carries its
|
|
365
|
+
* own endpoint and one-shot credential, while a request that names no endpoint
|
|
366
|
+
* is answered from this route's configuration. Capacities come from the
|
|
367
|
+
* fallback catalog when it names the id, so an adopting surface receives facts
|
|
368
|
+
* rather than having to invent them.
|
|
369
|
+
*
|
|
370
|
+
* @param {object} [request] - the draft being interrogated.
|
|
371
|
+
* @param {string} [request.provider] - route the draft edits, when it edits one.
|
|
372
|
+
* @param {string} [request.baseURL] - endpoint to interrogate.
|
|
373
|
+
* @param {string} [request.api] - protocol the draft names.
|
|
374
|
+
* @param {string} [request.apiKey] - credential for this interrogation alone.
|
|
375
|
+
* @param {AbortSignal} [signal] - caller cancellation.
|
|
376
|
+
* @returns {Promise<object[]>} the advertised models, in endpoint order.
|
|
377
|
+
* @throws {LlmError} when the request cannot be served or the endpoint refuses.
|
|
378
|
+
*/
|
|
379
|
+
const probe = async (request = {}, signal) => {
|
|
380
|
+
const provider = label(request.provider)
|
|
381
|
+
const endpoint = label(request.baseURL)
|
|
382
|
+
const api = label(request.api)
|
|
383
|
+
if (api !== undefined && api !== PROTOCOL) {
|
|
384
|
+
throw new LlmError(`llm-aliyun: this route speaks ${PROTOCOL}, not "${api}", and cannot list models for it`, 'DISCOVERY_UNSUPPORTED')
|
|
385
|
+
}
|
|
386
|
+
if (endpoint === undefined && provider !== undefined && provider !== 'aliyun') {
|
|
387
|
+
throw new LlmError(`llm-aliyun: the draft names route "${provider}", which this plugin does not own; give its endpoint to interrogate it`, 'DISCOVERY_UNSUPPORTED')
|
|
388
|
+
}
|
|
389
|
+
const supplied = label(request.apiKey)
|
|
390
|
+
const apiKey = supplied === undefined ? await resolveApiKey() : supplied
|
|
391
|
+
const listing = await fetchListing({
|
|
392
|
+
baseURL: endpoint ?? baseURL,
|
|
393
|
+
...apiKey === undefined || apiKey.length === 0 ? {} : { apiKey: assertUsableApiKey(apiKey, 'llm-aliyun', apiKeyEnv) },
|
|
394
|
+
headers,
|
|
395
|
+
timeoutMs,
|
|
396
|
+
signal,
|
|
397
|
+
fetchImpl,
|
|
398
|
+
})
|
|
399
|
+
const known = new Map(fallbackModels.map((entry) => [entry.id, entry]))
|
|
400
|
+
return listing.map((model) => {
|
|
401
|
+
const entry = known.get(model.id)
|
|
402
|
+
const contextWindow = model.contextWindow ?? entry?.contextWindow
|
|
403
|
+
const maxTokens = model.maxTokens ?? entry?.maxTokens
|
|
404
|
+
const input = entry?.input
|
|
405
|
+
return {
|
|
406
|
+
id: model.id,
|
|
407
|
+
name: model.name,
|
|
408
|
+
...contextWindow === undefined ? {} : { contextWindow },
|
|
409
|
+
...maxTokens === undefined ? {} : { maxTokens },
|
|
410
|
+
...input === undefined ? {} : { inputModalities: [...input] },
|
|
411
|
+
}
|
|
412
|
+
})
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
return {
|
|
416
|
+
/** Whether reads refresh at all. */
|
|
417
|
+
get enabled() {
|
|
418
|
+
return enabled
|
|
419
|
+
},
|
|
420
|
+
/** The ids in effect, or `undefined` before any listing. */
|
|
421
|
+
ids: () => ids,
|
|
422
|
+
/** When the listing in effect was read, or `0`. */
|
|
423
|
+
fetchedAt: () => fetchedAt,
|
|
424
|
+
/** The last failure message, cleared by a successful read. */
|
|
425
|
+
error: () => failure,
|
|
426
|
+
refresh,
|
|
427
|
+
ensureFresh,
|
|
428
|
+
settle,
|
|
429
|
+
probe,
|
|
430
|
+
}
|
|
431
|
+
}
|
package/lib/index.js
CHANGED
|
@@ -2,9 +2,15 @@
|
|
|
2
2
|
* Aliyun DashScope (Bailian) as a DeepSeek Harness model provider.
|
|
3
3
|
*
|
|
4
4
|
* This plugin registers one provider route, `aliyun`, on the harness LLM seam.
|
|
5
|
-
* It owns the route outright — endpoint, credential reference, protocol, and
|
|
6
|
-
*
|
|
7
|
-
*
|
|
5
|
+
* It owns the route outright — endpoint, credential reference, protocol, and the
|
|
6
|
+
* model catalog — so a profile needs no model list, and `pnpm update` is how the
|
|
7
|
+
* package moves forward.
|
|
8
|
+
*
|
|
9
|
+
* The catalog it advertises is live rather than shipped: {@link ./discovery.js}
|
|
10
|
+
* asks the configured endpoint which models it serves, {@link ./catalog.js}
|
|
11
|
+
* supplies the fallback membership and the model facts, and {@link ./models.js}
|
|
12
|
+
* decides what those two add up to. A profile can still pin the whole thing by
|
|
13
|
+
* setting `models`, and `discovery.enabled: false` turns the live half off.
|
|
8
14
|
*
|
|
9
15
|
* @module @doitian/dsh-provider-aliyun
|
|
10
16
|
*/
|
|
@@ -13,12 +19,22 @@ import z from '@deepseek-ai/schemastery'
|
|
|
13
19
|
|
|
14
20
|
import { createAliyunAdapter } from './adapter.js'
|
|
15
21
|
import {
|
|
16
|
-
|
|
22
|
+
CHAT_MODEL_PATTERNS,
|
|
17
23
|
DEFAULT_API_KEY_ENV,
|
|
18
24
|
DEFAULT_BASE_URL,
|
|
25
|
+
DISCOVERY_DEFAULTS,
|
|
26
|
+
DISCOVERY_TTL_MS,
|
|
27
|
+
DISCOVERY_WAIT_MS,
|
|
19
28
|
DISPLAY_NAME,
|
|
29
|
+
FALLBACK_MODELS,
|
|
30
|
+
NON_CHAT_MODEL_PATTERNS,
|
|
20
31
|
ROUTE,
|
|
21
32
|
} from './catalog.js'
|
|
33
|
+
import { DEFAULT_DISCOVERY_TIMEOUT_MS, createModelDiscovery } from './discovery.js'
|
|
34
|
+
// A build-time snapshot of Alibaba model facts; `npm run generate:metadata`
|
|
35
|
+
// refreshes it. Nothing at run time talks to the registry it came from.
|
|
36
|
+
import metadataSnapshot from './metadata.json' with { type: 'json' }
|
|
37
|
+
import { compilePatterns, createMetadataIndex, mergeCatalog, selectIds } from './models.js'
|
|
22
38
|
|
|
23
39
|
/** Entry name, used for logging and as the settings namespace fallback. */
|
|
24
40
|
export const name = 'llm-aliyun'
|
|
@@ -48,16 +64,29 @@ const modelSchema = z.object({
|
|
|
48
64
|
*
|
|
49
65
|
* Every field defaults to something serviceable, so the bundle's patch mounts
|
|
50
66
|
* this plugin with an empty config and the route works as soon as a key
|
|
51
|
-
* resolves. `models`
|
|
52
|
-
*
|
|
67
|
+
* resolves. `models` is the fallback catalog *and* the metadata table: discovery
|
|
68
|
+
* decides which ids the route advertises, and an id named here keeps the
|
|
69
|
+
* capacities written here.
|
|
53
70
|
*/
|
|
54
71
|
export const Config = z.object({
|
|
55
72
|
displayName: z.string().default(DISPLAY_NAME),
|
|
56
73
|
apiKeyEnv: z.string().role('credential-ref').default(DEFAULT_API_KEY_ENV),
|
|
57
74
|
baseURL: z.string().default(DEFAULT_BASE_URL),
|
|
58
|
-
models: z.array(modelSchema).default(
|
|
75
|
+
models: z.array(modelSchema).default(FALLBACK_MODELS),
|
|
59
76
|
reasoning: z.union(THINKING_LEVELS),
|
|
60
77
|
headers: z.dict(z.string()),
|
|
78
|
+
discovery: z.object({
|
|
79
|
+
enabled: z.boolean().default(true),
|
|
80
|
+
filter: z.union(['chat', 'patterns', 'all']).default('chat'),
|
|
81
|
+
include: z.array(z.string()).default([...CHAT_MODEL_PATTERNS]),
|
|
82
|
+
exclude: z.array(z.string()).default([...NON_CHAT_MODEL_PATTERNS]),
|
|
83
|
+
ttlMs: z.number().step(1).min(0).default(DISCOVERY_TTL_MS),
|
|
84
|
+
waitMs: z.number().step(1).min(0).default(DISCOVERY_WAIT_MS),
|
|
85
|
+
timeoutMs: z.number().step(1).min(1).default(DEFAULT_DISCOVERY_TIMEOUT_MS),
|
|
86
|
+
contextWindow: z.number().step(1).min(1).default(DISCOVERY_DEFAULTS.contextWindow),
|
|
87
|
+
maxTokens: z.number().step(1).min(1).default(DISCOVERY_DEFAULTS.maxTokens),
|
|
88
|
+
input: z.array(z.union(MODALITIES)).default([...DISCOVERY_DEFAULTS.input]),
|
|
89
|
+
}),
|
|
61
90
|
})
|
|
62
91
|
|
|
63
92
|
/**
|
|
@@ -96,15 +125,40 @@ export function apply(ctx, config) {
|
|
|
96
125
|
// The settings namespace is the entry id, so the Models page addresses this
|
|
97
126
|
// plugin's own row rather than any other adapter's section.
|
|
98
127
|
const settingsNs = ctx.fiber?.entry?.options.id ?? name
|
|
128
|
+
const discoveryConfig = config.discovery
|
|
129
|
+
const fallback = config.models
|
|
130
|
+
const resolveApiKey = apiKeyResolver(ctx)
|
|
131
|
+
|
|
132
|
+
// A workspace listing is the account's whole catalogue, so what it is allowed
|
|
133
|
+
// to advertise is filtered before it becomes a catalog. Two signals decide
|
|
134
|
+
// that — the profile's patterns and the metadata snapshot's own statement that
|
|
135
|
+
// a model answers with text — and a pattern that does not compile costs itself
|
|
136
|
+
// and nothing else.
|
|
137
|
+
const metadata = createMetadataIndex(metadataSnapshot)
|
|
138
|
+
const complain = (field) => (source, error) => {
|
|
139
|
+
ctx.logger?.warn?.(`llm-aliyun: discovery.${field} pattern ${JSON.stringify(source)} is not a valid regular expression (${error.message}); ignoring it`)
|
|
140
|
+
}
|
|
141
|
+
const filter = {
|
|
142
|
+
mode: discoveryConfig.filter,
|
|
143
|
+
include: compilePatterns(discoveryConfig.include, complain('include')),
|
|
144
|
+
exclude: compilePatterns(discoveryConfig.exclude, complain('exclude')),
|
|
145
|
+
// Naming a model in `models` is a stronger statement than any pattern.
|
|
146
|
+
pinned: fallback.map((entry) => entry.id),
|
|
147
|
+
known: (id) => metadata.knows(id),
|
|
148
|
+
}
|
|
149
|
+
ctx.logger?.debug?.(`llm-aliyun: model metadata snapshot from ${metadataSnapshot.source} (${metadata.size} models, generated ${metadataSnapshot.generatedAt})`)
|
|
99
150
|
|
|
100
|
-
|
|
151
|
+
// The adapter is built first only because discovery's change callback needs
|
|
152
|
+
// `setCatalog`; its own discovery callbacks are closures that run later.
|
|
153
|
+
const { adapter, setCatalog } = createAliyunAdapter({
|
|
101
154
|
displayName: config.displayName,
|
|
102
155
|
apiKeyEnv: config.apiKeyEnv,
|
|
103
156
|
baseURL: config.baseURL,
|
|
104
|
-
|
|
157
|
+
// What the route advertises until a listing arrives.
|
|
158
|
+
models: fallback,
|
|
105
159
|
reasoning: config.reasoning,
|
|
106
160
|
headers: config.headers,
|
|
107
|
-
resolveApiKey
|
|
161
|
+
resolveApiKey,
|
|
108
162
|
resolveAttachments: () => ctx.get('attachments'),
|
|
109
163
|
resolveImageAccess: (attachments, ref) => resolveImageAttachmentAccess(
|
|
110
164
|
attachments,
|
|
@@ -114,15 +168,55 @@ export function apply(ctx, config) {
|
|
|
114
168
|
onReplayDegrade: ({ provider, model, reason }) => {
|
|
115
169
|
ctx.logger?.warn?.(`llm-aliyun: unusable replay state on assistant history for route "${provider}/${model}"; sending that message as provider-neutral content (${reason})`)
|
|
116
170
|
},
|
|
171
|
+
// A read is the moment a stale listing is noticed, and a model list is the
|
|
172
|
+
// one read allowed to wait for the endpoint's answer.
|
|
173
|
+
onRead: () => discovery.ensureFresh(),
|
|
174
|
+
settle: () => discovery.settle({ maxWaitMs: discoveryConfig.waitMs }),
|
|
175
|
+
})
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Advertise a new listing, and tell the surfaces that read the old one.
|
|
179
|
+
*
|
|
180
|
+
* A model picker caches the catalog it read and re-reads it on
|
|
181
|
+
* `llm/adapters-updated` — the payload-free event the LLM seam publishes when
|
|
182
|
+
* a route set changes. Swapping the catalog alone is invisible to it, so the
|
|
183
|
+
* picker would keep showing whatever it read at startup until something else
|
|
184
|
+
* moved. Re-registering the same route is the documented way to publish that
|
|
185
|
+
* event: it changes no route, only the topology version the event announces.
|
|
186
|
+
*/
|
|
187
|
+
const announceCatalog = (models) => {
|
|
188
|
+
setCatalog(models)
|
|
189
|
+
registration.replace([ROUTE])
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
const discovery = createModelDiscovery({
|
|
193
|
+
baseURL: config.baseURL,
|
|
194
|
+
apiKeyEnv: config.apiKeyEnv,
|
|
195
|
+
enabled: discoveryConfig.enabled,
|
|
196
|
+
ttlMs: discoveryConfig.ttlMs,
|
|
197
|
+
timeoutMs: discoveryConfig.timeoutMs,
|
|
198
|
+
headers: config.headers,
|
|
199
|
+
fallbackModels: fallback,
|
|
200
|
+
resolveApiKey: () => resolveApiKey(ROUTE, { apiKeyEnv: config.apiKeyEnv }),
|
|
201
|
+
onChange: (ids) => announceCatalog(mergeCatalog({
|
|
202
|
+
ids: selectIds(ids, filter),
|
|
203
|
+
fallback,
|
|
204
|
+
defaults: {
|
|
205
|
+
contextWindow: discoveryConfig.contextWindow,
|
|
206
|
+
maxTokens: discoveryConfig.maxTokens,
|
|
207
|
+
input: discoveryConfig.input,
|
|
208
|
+
},
|
|
209
|
+
metadata,
|
|
210
|
+
})),
|
|
211
|
+
onError: (message) => ctx.logger?.warn?.(message),
|
|
117
212
|
})
|
|
118
213
|
|
|
119
214
|
// Both registrations are disposed with the plugin's fiber.
|
|
120
|
-
ctx.llm.registerAdapter([ROUTE], adapter)
|
|
215
|
+
const registration = ctx.llm.registerAdapter([ROUTE], adapter)
|
|
121
216
|
|
|
122
|
-
// The directory entry is what gives the route a row on the Models page
|
|
123
|
-
//
|
|
124
|
-
//
|
|
125
|
-
// an `aliyun` route through `llm-pi-ai`; remove that route to use this plugin.
|
|
217
|
+
// The directory entry is what gives the route a row on the Models page. It is
|
|
218
|
+
// refused if another adapter already declares `aliyun` — which is what happens
|
|
219
|
+
// when a profile still configures an `aliyun` route through `llm-pi-ai`.
|
|
126
220
|
ctx.llm.registerConfigurableProviders([{
|
|
127
221
|
provider: ROUTE,
|
|
128
222
|
displayName: config.displayName,
|
|
@@ -130,4 +224,16 @@ export function apply(ctx, config) {
|
|
|
130
224
|
settingsPath: [],
|
|
131
225
|
declared: true,
|
|
132
226
|
}])
|
|
227
|
+
|
|
228
|
+
// A configuration surface can interrogate this route's endpoint — even before
|
|
229
|
+
// a key is stored, by passing the draft's own.
|
|
230
|
+
ctx.llm.registerModelDiscovery(settingsNs, (request, signal) => discovery.probe(request, signal))
|
|
231
|
+
|
|
232
|
+
// The key can arrive at any time, and a stored key is exactly the event that
|
|
233
|
+
// makes a failed listing attempt worth retrying immediately.
|
|
234
|
+
ctx.on('credentials/reference-updated', (ref) => {
|
|
235
|
+
if (typeof ref === 'string' && ref === config.apiKeyEnv) discovery.ensureFresh({ force: true })
|
|
236
|
+
})
|
|
237
|
+
|
|
238
|
+
discovery.ensureFresh()
|
|
133
239
|
}
|