@siliconflow-official/dsh-llm-siliconflow 0.1.0-rc.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +22 -0
- package/README.md +168 -0
- package/cordis.patch.yml +10 -0
- package/lib/bin.js +219 -0
- package/lib/index.js +2 -0
- package/lib/invariant.js +23 -0
- package/lib/types/adapter.d.ts +114 -0
- package/lib/types/bin.d.ts +9 -0
- package/lib/types/discovery.d.ts +51 -0
- package/lib/types/index.d.ts +78 -0
- package/lib/types/invariant.d.ts +16 -0
- package/lib/types/serialize.d.ts +30 -0
- package/lib/types/setup.d.ts +102 -0
- package/lib/types/sse.d.ts +25 -0
- package/lib/types/translate.d.ts +37 -0
- package/lib/types/types.d.ts +145 -0
- package/lib/types-Bim1q3qs.js +937 -0
- package/package.json +73 -0
|
@@ -0,0 +1,937 @@
|
|
|
1
|
+
import z from "@deepseek-ai/schemastery";
|
|
2
|
+
import { CONTEXT_WINDOW_EXCEEDED_CODE, CallId, EMPTY_RESPONSE_CODE, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, RetryPolicySchema, assertUsableApiKey, attributionHeaders, contentHasImage, isContextWindowExceededError, isQuotaExceededError, resolveRetryPolicy } from "@deepseek-ai/dsh-llm";
|
|
3
|
+
import { credentialRef } from "@deepseek-ai/dsh-credentials";
|
|
4
|
+
import { launchEnvironmentOf } from "@deepseek-ai/dsh-launch-environment";
|
|
5
|
+
import { deepEqualJson, installSettingsSection, settingsNamespace } from "@deepseek-ai/dsh-settings";
|
|
6
|
+
import { MAX_TIMER_DELAY_MS, idleWatchdog, timeoutOf } from "@deepseek-ai/dsh-timeout";
|
|
7
|
+
import { getOrCreateAnonymousUserId } from "@deepseek-ai/dsh-anonymous-user-id";
|
|
8
|
+
import { EventSourceParserStream } from "eventsource-parser/stream";
|
|
9
|
+
//#region lib/types/discovery.js
|
|
10
|
+
/**
|
|
11
|
+
* Interrogate the SiliconFlow (OpenAI-compatible) `GET /models` listing for
|
|
12
|
+
* the chat models an endpoint serves, filtered with `sub_type=chat` and kept
|
|
13
|
+
* in the endpoint's own order — SiliconFlow's listing is already ordered by
|
|
14
|
+
* its own preference, so the adapter must not re-sort it.
|
|
15
|
+
*
|
|
16
|
+
* This module is transport-only: it takes the endpoint and bearer token for
|
|
17
|
+
* one interrogation and returns the entries it read. The registering plugin
|
|
18
|
+
* owns credential policy, and the adapter owns caching and the fallback to its
|
|
19
|
+
* configured catalog.
|
|
20
|
+
*
|
|
21
|
+
* @module dsh-llm-siliconflow/discovery
|
|
22
|
+
*/
|
|
23
|
+
/**
|
|
24
|
+
* Endpoint replies larger than this are refused. The bound holds on the bytes
|
|
25
|
+
* actually read, not the length the server claims, so a streaming or
|
|
26
|
+
* under-declaring reply cannot exhaust memory before it is turned away.
|
|
27
|
+
*/
|
|
28
|
+
const MAX_RESPONSE_BYTES = 4194304;
|
|
29
|
+
/** A positive integer field, or `undefined` when absent or unusable. */
|
|
30
|
+
function capacity(...candidates) {
|
|
31
|
+
for (const candidate of candidates) if (typeof candidate === "number" && Number.isInteger(candidate) && candidate > 0) return candidate;
|
|
32
|
+
}
|
|
33
|
+
/** A non-empty string field, or `undefined`. */
|
|
34
|
+
function label(...candidates) {
|
|
35
|
+
for (const candidate of candidates) if (typeof candidate === "string" && candidate.length > 0) return candidate;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Join the endpoint base with the chat-filtered listing path. The base is a
|
|
39
|
+
* prefix, not a URL to resolve against, so a gateway path such as
|
|
40
|
+
* `https://gateway.example/openai/v1` keeps its segments.
|
|
41
|
+
* @param baseURL - the chat-completions base.
|
|
42
|
+
* @returns the `GET /models?sub_type=chat` URL.
|
|
43
|
+
*/
|
|
44
|
+
function listingUrl(baseURL) {
|
|
45
|
+
return `${baseURL.replace(/\/+$/, "")}/models?sub_type=chat`;
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* Read a reply body, refusing one that outgrows {@link MAX_RESPONSE_BYTES}.
|
|
49
|
+
* @param response - the settled listing response.
|
|
50
|
+
* @param url - the endpoint, for the oversize diagnostic.
|
|
51
|
+
* @returns the decoded body text.
|
|
52
|
+
*/
|
|
53
|
+
async function readBounded(response, url) {
|
|
54
|
+
const oversized = () => new LlmError(`${url} answered with more than ${MAX_RESPONSE_BYTES} bytes`, "DISCOVERY_FAILED");
|
|
55
|
+
const declared = Number(response.headers.get("content-length") ?? NaN);
|
|
56
|
+
if (Number.isFinite(declared) && declared > MAX_RESPONSE_BYTES) {
|
|
57
|
+
await response.body?.cancel();
|
|
58
|
+
throw oversized();
|
|
59
|
+
}
|
|
60
|
+
/* v8 ignore next -- fetch always exposes a body stream on a 2xx Response; the null guard is defensive. */
|
|
61
|
+
if (response.body === null) return "";
|
|
62
|
+
const reader = response.body.getReader();
|
|
63
|
+
const chunks = [];
|
|
64
|
+
let total = 0;
|
|
65
|
+
try {
|
|
66
|
+
for (;;) {
|
|
67
|
+
const { done, value } = await reader.read();
|
|
68
|
+
if (done) break;
|
|
69
|
+
total += value.byteLength;
|
|
70
|
+
if (total > MAX_RESPONSE_BYTES) throw oversized();
|
|
71
|
+
chunks.push(value);
|
|
72
|
+
}
|
|
73
|
+
} finally {
|
|
74
|
+
/* v8 ignore next 4 -- cancel() after a completed or abandoned read settles without rejecting; unobserved best-effort cleanup. */
|
|
75
|
+
await reader.cancel().catch(() => {});
|
|
76
|
+
}
|
|
77
|
+
const body = new Uint8Array(total);
|
|
78
|
+
let offset = 0;
|
|
79
|
+
for (const chunk of chunks) {
|
|
80
|
+
body.set(chunk, offset);
|
|
81
|
+
offset += chunk.byteLength;
|
|
82
|
+
}
|
|
83
|
+
return new TextDecoder().decode(body);
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* Map one listing reply into entries, preserving endpoint order. A row without
|
|
87
|
+
* a usable id is skipped rather than failing the whole interrogation: a single
|
|
88
|
+
* malformed row should not hide the rest of a working endpoint's catalog.
|
|
89
|
+
* @param body - the parsed reply body.
|
|
90
|
+
* @returns the entries in arrival order.
|
|
91
|
+
*/
|
|
92
|
+
function readListing(body) {
|
|
93
|
+
const data = body?.data;
|
|
94
|
+
if (!Array.isArray(data)) throw new LlmError("the endpoint's model listing has no \"data\" array", "DISCOVERY_FAILED");
|
|
95
|
+
const entries = [];
|
|
96
|
+
for (const raw of data) {
|
|
97
|
+
const entry = raw;
|
|
98
|
+
const id = label(entry?.id);
|
|
99
|
+
if (id === void 0) continue;
|
|
100
|
+
const name = label(entry?.name, entry?.display_name);
|
|
101
|
+
const contextWindow = capacity(entry?.context_window, entry?.context_length);
|
|
102
|
+
const maxTokens = capacity(entry?.max_output_tokens, entry?.max_tokens);
|
|
103
|
+
entries.push({
|
|
104
|
+
id,
|
|
105
|
+
...name === void 0 ? {} : { name },
|
|
106
|
+
...contextWindow === void 0 ? {} : { contextWindow },
|
|
107
|
+
...maxTokens === void 0 ? {} : { maxTokens }
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
return entries;
|
|
111
|
+
}
|
|
112
|
+
/**
|
|
113
|
+
* Interrogate one endpoint for the chat models it advertises.
|
|
114
|
+
* @param baseURL - the chat-completions base; `/models?sub_type=chat` is appended.
|
|
115
|
+
* @param apiKey - bearer token, or `undefined` to probe unauthenticated.
|
|
116
|
+
* @param signal - caller cancellation; the fetch and body read honor it.
|
|
117
|
+
* @returns the advertised models in endpoint order.
|
|
118
|
+
* @throws LlmError when the endpoint is unreachable, refuses the request, or
|
|
119
|
+
* the reply is not a readable listing.
|
|
120
|
+
*/
|
|
121
|
+
async function discoverChatModels(baseURL, apiKey, signal) {
|
|
122
|
+
const url = listingUrl(baseURL);
|
|
123
|
+
let response;
|
|
124
|
+
try {
|
|
125
|
+
response = await fetch(url, {
|
|
126
|
+
method: "GET",
|
|
127
|
+
headers: {
|
|
128
|
+
accept: "application/json",
|
|
129
|
+
...apiKey === void 0 ? {} : { authorization: `Bearer ${apiKey}` },
|
|
130
|
+
...attributionHeaders()
|
|
131
|
+
},
|
|
132
|
+
...signal === void 0 ? {} : { signal }
|
|
133
|
+
});
|
|
134
|
+
} catch (error) {
|
|
135
|
+
if (signal?.aborted) throw new LlmError("model discovery aborted by caller", "ABORTED", { cause: error });
|
|
136
|
+
throw new LlmError(`could not reach ${url}`, "DISCOVERY_FAILED", { cause: error });
|
|
137
|
+
}
|
|
138
|
+
if (!response.ok) throw new LlmError(`${url} answered ${response.status}`, "DISCOVERY_FAILED");
|
|
139
|
+
let text;
|
|
140
|
+
try {
|
|
141
|
+
text = await readBounded(response, url);
|
|
142
|
+
} catch (error) {
|
|
143
|
+
if (signal?.aborted) throw new LlmError("model discovery aborted by caller", "ABORTED", { cause: error });
|
|
144
|
+
throw error;
|
|
145
|
+
}
|
|
146
|
+
let body;
|
|
147
|
+
try {
|
|
148
|
+
body = JSON.parse(text);
|
|
149
|
+
} catch (error) {
|
|
150
|
+
throw new LlmError(`${url} did not answer with JSON`, "DISCOVERY_FAILED", { cause: error });
|
|
151
|
+
}
|
|
152
|
+
return readListing(body);
|
|
153
|
+
}
|
|
154
|
+
//#endregion
|
|
155
|
+
//#region lib/types/serialize.js
|
|
156
|
+
/**
|
|
157
|
+
* Serialize harness messages into SiliconFlow chat completions. User text is
|
|
158
|
+
* joined; assistant text becomes `content`, tool calls become `tool_calls`,
|
|
159
|
+
* and tool results become separate tool messages. Assistant reasoning is
|
|
160
|
+
* replayed as `reasoning_content` only on tool-call turns, as hosted reasoning
|
|
161
|
+
* models (DeepSeek-R1 and siblings) require. Core image blocks are rejected
|
|
162
|
+
* explicitly because this wire route is text-only; unknown declaration-merged
|
|
163
|
+
* block types retain the adapter's documented extension fallback.
|
|
164
|
+
* @module dsh-llm-siliconflow/serialize
|
|
165
|
+
*/
|
|
166
|
+
/** Join the text blocks of a message (used for user/tool-result content). */
|
|
167
|
+
function flattenText(blocks) {
|
|
168
|
+
return blocks.filter((block) => block.type === "text").map((block) => block.text).join("");
|
|
169
|
+
}
|
|
170
|
+
/** Reject core image content before any text-flattening path can silently erase it. */
|
|
171
|
+
function assertTextOnly(blocks) {
|
|
172
|
+
if (contentHasImage(blocks)) throw new LlmError("The SiliconFlow chat-completions adapter does not support image content.", "UNSUPPORTED_CONTENT");
|
|
173
|
+
}
|
|
174
|
+
/** Serialize one assistant message (text + reasoning + tool calls). */
|
|
175
|
+
function serializeAssistant(message) {
|
|
176
|
+
const text = flattenText(message.content);
|
|
177
|
+
const reasoning = message.content.filter((block) => block.type === "reasoning").map((block) => block.text).join("");
|
|
178
|
+
const toolCalls = message.content.filter((block) => block.type === "tool-call").map((block) => ({
|
|
179
|
+
id: block.id,
|
|
180
|
+
type: "function",
|
|
181
|
+
function: {
|
|
182
|
+
name: block.name,
|
|
183
|
+
arguments: block.arguments
|
|
184
|
+
}
|
|
185
|
+
}));
|
|
186
|
+
return {
|
|
187
|
+
role: "assistant",
|
|
188
|
+
content: text,
|
|
189
|
+
...toolCalls.length > 0 && reasoning.length > 0 ? { reasoning_content: reasoning } : {},
|
|
190
|
+
...toolCalls.length > 0 ? { tool_calls: toolCalls } : {}
|
|
191
|
+
};
|
|
192
|
+
}
|
|
193
|
+
/**
|
|
194
|
+
* Serialize the conversation. `tool-result` blocks become standalone
|
|
195
|
+
* `{role: 'tool'}` messages; the harness puts each tool result in its own
|
|
196
|
+
* user-role message, so a mixed user message contributes its text first and
|
|
197
|
+
* its tool results as separate wire messages after.
|
|
198
|
+
* @param messages - the harness conversation, in order.
|
|
199
|
+
* @returns the wire messages; order preserved, each tool result expanded into its own entry.
|
|
200
|
+
*/
|
|
201
|
+
function serializeMessages(messages) {
|
|
202
|
+
const wire = [];
|
|
203
|
+
for (const message of messages) {
|
|
204
|
+
assertTextOnly(message.content);
|
|
205
|
+
if (message.role === "system") {
|
|
206
|
+
wire.push({
|
|
207
|
+
role: "system",
|
|
208
|
+
content: flattenText(message.content)
|
|
209
|
+
});
|
|
210
|
+
continue;
|
|
211
|
+
}
|
|
212
|
+
if (message.role === "assistant") {
|
|
213
|
+
wire.push(serializeAssistant(message));
|
|
214
|
+
continue;
|
|
215
|
+
}
|
|
216
|
+
const toolResults = message.content.filter((block) => block.type === "tool-result");
|
|
217
|
+
const text = flattenText(message.content);
|
|
218
|
+
if (text.length > 0 || toolResults.length === 0) wire.push({
|
|
219
|
+
role: "user",
|
|
220
|
+
content: text
|
|
221
|
+
});
|
|
222
|
+
for (const result of toolResults) wire.push({
|
|
223
|
+
role: "tool",
|
|
224
|
+
tool_call_id: result.toolCallId,
|
|
225
|
+
content: flattenText(result.content) || "(no output)"
|
|
226
|
+
});
|
|
227
|
+
}
|
|
228
|
+
return wire;
|
|
229
|
+
}
|
|
230
|
+
/**
|
|
231
|
+
* Build the full wire request. Always streaming (`stream: true`, usage
|
|
232
|
+
* reporting on); optional fields are omitted rather than sent as null, so
|
|
233
|
+
* provider defaults apply.
|
|
234
|
+
* @param options - the harness request (model, history, system, tools, sampling).
|
|
235
|
+
* @returns the chat-completions request body.
|
|
236
|
+
*/
|
|
237
|
+
function serializeRequest(options) {
|
|
238
|
+
const messages = [];
|
|
239
|
+
if (options.system !== void 0) messages.push({
|
|
240
|
+
role: "system",
|
|
241
|
+
content: options.system
|
|
242
|
+
});
|
|
243
|
+
messages.push(...serializeMessages(options.messages));
|
|
244
|
+
const tools = options.tools?.map((tool) => ({
|
|
245
|
+
type: "function",
|
|
246
|
+
function: {
|
|
247
|
+
name: tool.name,
|
|
248
|
+
description: tool.description,
|
|
249
|
+
parameters: tool.parameters
|
|
250
|
+
}
|
|
251
|
+
}));
|
|
252
|
+
return {
|
|
253
|
+
model: options.model,
|
|
254
|
+
messages,
|
|
255
|
+
stream: true,
|
|
256
|
+
stream_options: { include_usage: true },
|
|
257
|
+
...tools !== void 0 && tools.length > 0 ? { tools } : {},
|
|
258
|
+
...options.temperature !== void 0 ? { temperature: options.temperature } : {},
|
|
259
|
+
...options.maxTokens === void 0 ? {} : { max_tokens: options.maxTokens },
|
|
260
|
+
...options.stop !== void 0 ? { stop: options.stop } : {}
|
|
261
|
+
};
|
|
262
|
+
}
|
|
263
|
+
/**
|
|
264
|
+
* Parse an SSE byte stream into data payloads. Yields `[DONE]` as the final
|
|
265
|
+
* value and returns; throws `LlmError('STREAM_CLOSED')` when the stream ends
|
|
266
|
+
* without it (truncated response — the model call cannot be trusted).
|
|
267
|
+
* @param stream - raw SSE bytes; reads may split anywhere, including mid-UTF-8 sequence.
|
|
268
|
+
* @param onComment - optional transport-activity callback; comments never enter the yielded payload stream.
|
|
269
|
+
* @returns each event's data payload in arrival order, the `[DONE]` sentinel last.
|
|
270
|
+
*/
|
|
271
|
+
async function* parseSse(stream, onComment) {
|
|
272
|
+
const events = stream.pipeThrough(new TextDecoderStream()).pipeThrough(new EventSourceParserStream({ onComment }));
|
|
273
|
+
for await (const { data } of events) {
|
|
274
|
+
yield data;
|
|
275
|
+
if (data === "[DONE]") return;
|
|
276
|
+
}
|
|
277
|
+
throw new LlmError("SSE stream ended without [DONE]", "STREAM_CLOSED");
|
|
278
|
+
}
|
|
279
|
+
//#endregion
|
|
280
|
+
//#region lib/types/translate.js
|
|
281
|
+
/**
|
|
282
|
+
* Translate SiliconFlow SSE payloads with one stateful harness block per
|
|
283
|
+
* content, reasoning, or tool-call index. An empty initial reasoning delta
|
|
284
|
+
* does not open a block. Finish reason and the latest usage are deferred until
|
|
285
|
+
* `[DONE]`, covering both finish-attached and trailing usage-only shapes while
|
|
286
|
+
* ensuring no chunk follows `finish`.
|
|
287
|
+
*
|
|
288
|
+
* Translate SiliconFlow wire chunks into the harness `StreamChunk` protocol.
|
|
289
|
+
* @module dsh-llm-siliconflow/translate
|
|
290
|
+
*/
|
|
291
|
+
/**
|
|
292
|
+
* Map the wire finish_reason vocabulary to the harness FinishReason.
|
|
293
|
+
* @param reason - the wire `finish_reason` string.
|
|
294
|
+
* @returns the mapped reason; unrecognized values (content_filter, …) become `{kind: 'error'}` with the uppercased value as `code`.
|
|
295
|
+
*/
|
|
296
|
+
function mapFinishReason(reason) {
|
|
297
|
+
switch (reason) {
|
|
298
|
+
case "stop": return { kind: "stop" };
|
|
299
|
+
case "tool_calls": return { kind: "tool-calls" };
|
|
300
|
+
case "length": return { kind: "max-tokens" };
|
|
301
|
+
default: return {
|
|
302
|
+
kind: "error",
|
|
303
|
+
failure: {
|
|
304
|
+
message: `model stopped: ${reason}`,
|
|
305
|
+
code: reason.toUpperCase()
|
|
306
|
+
}
|
|
307
|
+
};
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
/**
|
|
311
|
+
* Map wire usage fields. SiliconFlow's `prompt_tokens` INCLUDES cache hits
|
|
312
|
+
* (`prompt_tokens = prompt_cache_hit_tokens + prompt_cache_miss_tokens`); the
|
|
313
|
+
* harness TokenUsage convention is DISJOINT counts, so cache reads are
|
|
314
|
+
* subtracted out of `inputTokens`.
|
|
315
|
+
* @param usage - wire usage from the finish chunk or the trailing usage-only chunk.
|
|
316
|
+
* @returns disjoint harness counts; cache/reasoning fields present only when the wire reported them.
|
|
317
|
+
*/
|
|
318
|
+
function mapUsage(usage) {
|
|
319
|
+
const cacheRead = usage.prompt_tokens_details?.cached_tokens ?? usage.prompt_cache_hit_tokens;
|
|
320
|
+
const reasoning = usage.completion_tokens_details?.reasoning_tokens;
|
|
321
|
+
return {
|
|
322
|
+
inputTokens: usage.prompt_tokens - (cacheRead ?? 0),
|
|
323
|
+
outputTokens: usage.completion_tokens,
|
|
324
|
+
...cacheRead !== void 0 ? { cacheReadTokens: cacheRead } : {},
|
|
325
|
+
...reasoning !== void 0 ? { reasoningTokens: reasoning } : {}
|
|
326
|
+
};
|
|
327
|
+
}
|
|
328
|
+
/** Assemble the final ContentBlock for one open block. */
|
|
329
|
+
function closeBlock(block) {
|
|
330
|
+
switch (block.kind) {
|
|
331
|
+
case "text": return {
|
|
332
|
+
type: "text",
|
|
333
|
+
text: block.text
|
|
334
|
+
};
|
|
335
|
+
case "reasoning": return {
|
|
336
|
+
type: "reasoning",
|
|
337
|
+
text: block.text
|
|
338
|
+
};
|
|
339
|
+
case "tool-call": return {
|
|
340
|
+
type: "tool-call",
|
|
341
|
+
id: CallId(block.callId ?? ""),
|
|
342
|
+
name: block.name ?? "",
|
|
343
|
+
arguments: block.text
|
|
344
|
+
};
|
|
345
|
+
}
|
|
346
|
+
}
|
|
347
|
+
/**
|
|
348
|
+
* Consume SSE data payloads (ending with `[DONE]`) and yield StreamChunks.
|
|
349
|
+
* Malformed JSON payloads abort the stream with `MALFORMED_RESPONSE`.
|
|
350
|
+
* @param payloads - SSE data payloads from {@link parseSse}, `[DONE]`-terminated.
|
|
351
|
+
* @returns deltas as they arrive; `block-end`s, `usage`, and `finish` are all deferred to the `[DONE]` sentinel.
|
|
352
|
+
* A `stop` (or absent) finish with no opened blocks is a degenerate provider completion and maps to an
|
|
353
|
+
* `EMPTY_RESPONSE` error finish instead of a successful empty message.
|
|
354
|
+
*/
|
|
355
|
+
async function* translate(payloads) {
|
|
356
|
+
let nextIndex = 0;
|
|
357
|
+
let textBlock;
|
|
358
|
+
let reasoningBlock;
|
|
359
|
+
const toolBlocks = /* @__PURE__ */ new Map();
|
|
360
|
+
const order = [];
|
|
361
|
+
let pendingFinish;
|
|
362
|
+
let pendingUsage;
|
|
363
|
+
function open(kind) {
|
|
364
|
+
const block = {
|
|
365
|
+
index: nextIndex++,
|
|
366
|
+
kind,
|
|
367
|
+
text: ""
|
|
368
|
+
};
|
|
369
|
+
order.push(block);
|
|
370
|
+
return block;
|
|
371
|
+
}
|
|
372
|
+
for await (const payload of payloads) {
|
|
373
|
+
if (payload === "[DONE]") {
|
|
374
|
+
for (const block of order) yield {
|
|
375
|
+
type: "block-end",
|
|
376
|
+
index: block.index,
|
|
377
|
+
block: closeBlock(block)
|
|
378
|
+
};
|
|
379
|
+
if (pendingUsage) yield {
|
|
380
|
+
type: "usage",
|
|
381
|
+
usage: pendingUsage
|
|
382
|
+
};
|
|
383
|
+
const reason = pendingFinish ?? { kind: "stop" };
|
|
384
|
+
yield {
|
|
385
|
+
type: "finish",
|
|
386
|
+
reason: reason.kind === "stop" && order.length === 0 ? {
|
|
387
|
+
kind: "error",
|
|
388
|
+
failure: {
|
|
389
|
+
message: "model returned a completed response with no content",
|
|
390
|
+
code: EMPTY_RESPONSE_CODE
|
|
391
|
+
}
|
|
392
|
+
} : reason
|
|
393
|
+
};
|
|
394
|
+
return;
|
|
395
|
+
}
|
|
396
|
+
let chunk;
|
|
397
|
+
try {
|
|
398
|
+
chunk = JSON.parse(payload);
|
|
399
|
+
} catch {
|
|
400
|
+
throw new LlmError(`malformed SSE payload: ${payload.slice(0, 120)}`, "MALFORMED_RESPONSE");
|
|
401
|
+
}
|
|
402
|
+
for (const choice of chunk.choices ?? []) {
|
|
403
|
+
const delta = choice.delta;
|
|
404
|
+
const reasoning = delta?.reasoning_content;
|
|
405
|
+
if (typeof reasoning === "string" && reasoning.length > 0) {
|
|
406
|
+
if (!reasoningBlock) {
|
|
407
|
+
reasoningBlock = open("reasoning");
|
|
408
|
+
yield {
|
|
409
|
+
type: "block-start",
|
|
410
|
+
index: reasoningBlock.index,
|
|
411
|
+
blockType: "reasoning"
|
|
412
|
+
};
|
|
413
|
+
}
|
|
414
|
+
reasoningBlock.text += reasoning;
|
|
415
|
+
yield {
|
|
416
|
+
type: "reasoning-delta",
|
|
417
|
+
index: reasoningBlock.index,
|
|
418
|
+
text: reasoning
|
|
419
|
+
};
|
|
420
|
+
}
|
|
421
|
+
const content = delta?.content;
|
|
422
|
+
if (typeof content === "string" && content.length > 0) {
|
|
423
|
+
if (!textBlock) {
|
|
424
|
+
textBlock = open("text");
|
|
425
|
+
yield {
|
|
426
|
+
type: "block-start",
|
|
427
|
+
index: textBlock.index,
|
|
428
|
+
blockType: "text"
|
|
429
|
+
};
|
|
430
|
+
}
|
|
431
|
+
textBlock.text += content;
|
|
432
|
+
yield {
|
|
433
|
+
type: "text-delta",
|
|
434
|
+
index: textBlock.index,
|
|
435
|
+
text: content
|
|
436
|
+
};
|
|
437
|
+
}
|
|
438
|
+
for (const call of delta?.tool_calls ?? []) {
|
|
439
|
+
let block = toolBlocks.get(call.index);
|
|
440
|
+
if (!block) {
|
|
441
|
+
block = open("tool-call");
|
|
442
|
+
toolBlocks.set(call.index, block);
|
|
443
|
+
yield {
|
|
444
|
+
type: "block-start",
|
|
445
|
+
index: block.index,
|
|
446
|
+
blockType: "tool-call"
|
|
447
|
+
};
|
|
448
|
+
}
|
|
449
|
+
if (call.id !== void 0) block.callId = call.id;
|
|
450
|
+
if (call.function?.name !== void 0) block.name = call.function.name;
|
|
451
|
+
const fragment = call.function?.arguments ?? "";
|
|
452
|
+
block.text += fragment;
|
|
453
|
+
yield {
|
|
454
|
+
type: "tool-call-delta",
|
|
455
|
+
index: block.index,
|
|
456
|
+
id: CallId(block.callId ?? ""),
|
|
457
|
+
...block.name !== void 0 ? { name: block.name } : {},
|
|
458
|
+
argumentsDelta: fragment
|
|
459
|
+
};
|
|
460
|
+
}
|
|
461
|
+
if (typeof choice.finish_reason === "string") pendingFinish = mapFinishReason(choice.finish_reason);
|
|
462
|
+
}
|
|
463
|
+
if (chunk.usage) pendingUsage = mapUsage(chunk.usage);
|
|
464
|
+
}
|
|
465
|
+
throw new LlmError("SSE payload stream ended without [DONE]", "STREAM_CLOSED");
|
|
466
|
+
}
|
|
467
|
+
//#endregion
|
|
468
|
+
//#region lib/types/adapter.js
|
|
469
|
+
/**
|
|
470
|
+
* `SiliconFlowAdapter`: fetch + SSE against a SiliconFlow (OpenAI-compatible)
|
|
471
|
+
* chat-completions endpoint, emitting harness StreamChunks. The adapter is
|
|
472
|
+
* transport-only: connection facts arrive through a thunk resolved once per
|
|
473
|
+
* operation and the bearer token through a per-request resolver, so the
|
|
474
|
+
* registering plugin owns validation, layering, and credential policy.
|
|
475
|
+
*
|
|
476
|
+
* @module dsh-llm-siliconflow/adapter
|
|
477
|
+
*/
|
|
478
|
+
var __addDisposableResource = function(env, value, async) {
|
|
479
|
+
if (value !== null && value !== void 0) {
|
|
480
|
+
if (typeof value !== "object" && typeof value !== "function") throw new TypeError("Object expected.");
|
|
481
|
+
var dispose, inner;
|
|
482
|
+
if (async) {
|
|
483
|
+
if (!Symbol.asyncDispose) throw new TypeError("Symbol.asyncDispose is not defined.");
|
|
484
|
+
dispose = value[Symbol.asyncDispose];
|
|
485
|
+
}
|
|
486
|
+
if (dispose === void 0) {
|
|
487
|
+
if (!Symbol.dispose) throw new TypeError("Symbol.dispose is not defined.");
|
|
488
|
+
dispose = value[Symbol.dispose];
|
|
489
|
+
if (async) inner = dispose;
|
|
490
|
+
}
|
|
491
|
+
if (typeof dispose !== "function") throw new TypeError("Object not disposable.");
|
|
492
|
+
if (inner) dispose = function() {
|
|
493
|
+
try {
|
|
494
|
+
inner.call(this);
|
|
495
|
+
} catch (e) {
|
|
496
|
+
return Promise.reject(e);
|
|
497
|
+
}
|
|
498
|
+
};
|
|
499
|
+
env.stack.push({
|
|
500
|
+
value,
|
|
501
|
+
dispose,
|
|
502
|
+
async
|
|
503
|
+
});
|
|
504
|
+
} else if (async) env.stack.push({ async: true });
|
|
505
|
+
return value;
|
|
506
|
+
};
|
|
507
|
+
var __disposeResources = (function(SuppressedError) {
|
|
508
|
+
return function(env) {
|
|
509
|
+
function fail(e) {
|
|
510
|
+
env.error = env.hasError ? new SuppressedError(e, env.error, "An error was suppressed during disposal.") : e;
|
|
511
|
+
env.hasError = true;
|
|
512
|
+
}
|
|
513
|
+
var r, s = 0;
|
|
514
|
+
function next() {
|
|
515
|
+
while (r = env.stack.pop()) try {
|
|
516
|
+
if (!r.async && s === 1) return s = 0, env.stack.push(r), Promise.resolve().then(next);
|
|
517
|
+
if (r.dispose) {
|
|
518
|
+
var result = r.dispose.call(r.value);
|
|
519
|
+
if (r.async) return s |= 2, Promise.resolve(result).then(next, function(e) {
|
|
520
|
+
fail(e);
|
|
521
|
+
return next();
|
|
522
|
+
});
|
|
523
|
+
} else s |= 1;
|
|
524
|
+
} catch (e) {
|
|
525
|
+
fail(e);
|
|
526
|
+
}
|
|
527
|
+
if (s === 1) return env.hasError ? Promise.reject(env.error) : Promise.resolve();
|
|
528
|
+
if (env.hasError) throw env.error;
|
|
529
|
+
}
|
|
530
|
+
return next();
|
|
531
|
+
};
|
|
532
|
+
})(typeof SuppressedError === "function" ? SuppressedError : function(error, suppressed, message) {
|
|
533
|
+
var e = new Error(message);
|
|
534
|
+
return e.name = "SuppressedError", e.error = error, e.suppressed = suppressed, e;
|
|
535
|
+
});
|
|
536
|
+
/** Default maximum idle interval while an adapter stream read is outstanding. */
|
|
537
|
+
const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 3e5;
|
|
538
|
+
/** Default combined request/response context capacity. */
|
|
539
|
+
const DEFAULT_CONTEXT_WINDOW = 32768;
|
|
540
|
+
/** Default per-request output-token cap. */
|
|
541
|
+
const DEFAULT_MAX_TOKENS = 8192;
|
|
542
|
+
/** How long a cached model-listing discovery stays fresh before the next `listModels` re-interrogates. */
|
|
543
|
+
const DISCOVERY_TTL_MS = 3e5;
|
|
544
|
+
const STREAM_IDLE_TIMEOUT_CODE = "LLM_STREAM_IDLE_TIMEOUT";
|
|
545
|
+
function modelInfo(provider, model) {
|
|
546
|
+
return {
|
|
547
|
+
provider,
|
|
548
|
+
id: model.id,
|
|
549
|
+
name: model.name ?? model.id,
|
|
550
|
+
...model.description === void 0 ? {} : { description: model.description },
|
|
551
|
+
inputModalities: ["text"]
|
|
552
|
+
};
|
|
553
|
+
}
|
|
554
|
+
function providerRetryAfterMs(value) {
|
|
555
|
+
if (value === null) return void 0;
|
|
556
|
+
if (/^\d+$/.test(value)) {
|
|
557
|
+
const delay = Number(value) * 1e3;
|
|
558
|
+
return Number.isFinite(delay) && delay > 0 ? delay : void 0;
|
|
559
|
+
}
|
|
560
|
+
const delay = Date.parse(value) - Date.now();
|
|
561
|
+
return Number.isFinite(delay) && delay > 0 ? delay : void 0;
|
|
562
|
+
}
|
|
563
|
+
function requestId(headers) {
|
|
564
|
+
const value = headers.get("x-request-id");
|
|
565
|
+
return value === null || value.length === 0 ? void 0 : ProviderRequestId(value);
|
|
566
|
+
}
|
|
567
|
+
/**
|
|
568
|
+
* Map an HTTP status to a stable LlmError code.
|
|
569
|
+
* @param status - status of a non-2xx provider response.
|
|
570
|
+
* @param error - parsed provider error body, when available.
|
|
571
|
+
* @returns the normalized harness error code.
|
|
572
|
+
*/
|
|
573
|
+
function httpErrorCode(status, error) {
|
|
574
|
+
if (status === 401 || status === 403) return "AUTH";
|
|
575
|
+
const detail = [
|
|
576
|
+
error?.code,
|
|
577
|
+
error?.type,
|
|
578
|
+
error?.message
|
|
579
|
+
].filter(Boolean).join(" ");
|
|
580
|
+
if (isQuotaExceededError(detail)) return QUOTA_EXCEEDED_CODE;
|
|
581
|
+
if (status === 429) return "RATE_LIMIT";
|
|
582
|
+
if (status === 400) {
|
|
583
|
+
if (isContextWindowExceededError(detail)) return CONTEXT_WINDOW_EXCEEDED_CODE;
|
|
584
|
+
return "INVALID_REQUEST";
|
|
585
|
+
}
|
|
586
|
+
if (status >= 500) return "SERVER";
|
|
587
|
+
return `HTTP_${status}`;
|
|
588
|
+
}
|
|
589
|
+
/**
|
|
590
|
+
* A direct-fetch `LlmAdapter` for SiliconFlow's OpenAI-compatible
|
|
591
|
+
* chat-completions endpoint. One instance serves every model name it was
|
|
592
|
+
* registered under (the harness model name IS the wire model name).
|
|
593
|
+
*
|
|
594
|
+
* One stable signal reaches both initial fetch and body reads. Caller aborts
|
|
595
|
+
* map to `ABORTED`; the configured per-read idle watchdog maps to `TIMEOUT`.
|
|
596
|
+
*/
|
|
597
|
+
var SiliconFlowAdapter = class extends LlmAdapter {
|
|
598
|
+
config;
|
|
599
|
+
/** Cached listing result, keyed by the baseURL it was read from. */
|
|
600
|
+
discovery;
|
|
601
|
+
constructor(config) {
|
|
602
|
+
super();
|
|
603
|
+
this.config = config;
|
|
604
|
+
}
|
|
605
|
+
providerInfo(provider) {
|
|
606
|
+
return {
|
|
607
|
+
id: provider,
|
|
608
|
+
name: "SiliconFlow"
|
|
609
|
+
};
|
|
610
|
+
}
|
|
611
|
+
providerRetryPolicy(_provider) {
|
|
612
|
+
return this.config.options().retryPolicy;
|
|
613
|
+
}
|
|
614
|
+
/** The cached listing for `connection` when it is still fresh; never re-interrogates. */
|
|
615
|
+
freshDiscovery(connection) {
|
|
616
|
+
if (this.discovery?.baseURL !== connection.baseURL) return void 0;
|
|
617
|
+
if (Date.now() - this.discovery.fetchedAt > 3e5) return void 0;
|
|
618
|
+
return this.discovery.models;
|
|
619
|
+
}
|
|
620
|
+
/**
|
|
621
|
+
* The models this adapter currently advertises: the live chat listing when a
|
|
622
|
+
* key and a reachable endpoint can supply one, else the configured catalog.
|
|
623
|
+
* Discovery is advisory and best-effort — a missing key or any interrogation
|
|
624
|
+
* failure falls back to the configured `models` rather than breaking the
|
|
625
|
+
* picker, because an absent catalog would hide the provider entirely.
|
|
626
|
+
*/
|
|
627
|
+
async discover(connection) {
|
|
628
|
+
const cached = this.freshDiscovery(connection);
|
|
629
|
+
if (cached !== void 0) return cached;
|
|
630
|
+
try {
|
|
631
|
+
const apiKey = await this.config.resolveApiKey(connection);
|
|
632
|
+
const models = await discoverChatModels(connection.baseURL, apiKey);
|
|
633
|
+
this.discovery = {
|
|
634
|
+
baseURL: connection.baseURL,
|
|
635
|
+
models: [...models],
|
|
636
|
+
fetchedAt: Date.now()
|
|
637
|
+
};
|
|
638
|
+
return this.discovery.models;
|
|
639
|
+
} catch {
|
|
640
|
+
this.discovery = void 0;
|
|
641
|
+
return connection.models;
|
|
642
|
+
}
|
|
643
|
+
}
|
|
644
|
+
async listModels(provider) {
|
|
645
|
+
const connection = this.config.options();
|
|
646
|
+
return (await this.discover(connection)).map((model) => modelInfo(provider, model));
|
|
647
|
+
}
|
|
648
|
+
resolveModel(provider, model, _signal) {
|
|
649
|
+
const connection = this.config.options();
|
|
650
|
+
const configured = connection.models.find((entry) => entry.id === model);
|
|
651
|
+
const discovered = this.freshDiscovery(connection)?.find((entry) => entry.id === model);
|
|
652
|
+
const entry = configured ?? discovered;
|
|
653
|
+
return Promise.resolve({
|
|
654
|
+
...entry === void 0 ? {
|
|
655
|
+
provider,
|
|
656
|
+
id: model,
|
|
657
|
+
name: model,
|
|
658
|
+
inputModalities: ["text"]
|
|
659
|
+
} : modelInfo(provider, entry),
|
|
660
|
+
context: { contextWindow: entry?.contextWindow ?? connection.defaultContextWindow },
|
|
661
|
+
defaultMaxTokens: entry?.maxTokens ?? connection.maxTokens
|
|
662
|
+
});
|
|
663
|
+
}
|
|
664
|
+
async *stream(options) {
|
|
665
|
+
const env_1 = {
|
|
666
|
+
stack: [],
|
|
667
|
+
error: void 0,
|
|
668
|
+
hasError: false
|
|
669
|
+
};
|
|
670
|
+
try {
|
|
671
|
+
const connection = this.config.options();
|
|
672
|
+
const apiKey = await this.config.resolveApiKey(connection);
|
|
673
|
+
const userId = this.config.resolveUserId();
|
|
674
|
+
const consumer = new AbortController();
|
|
675
|
+
const upstream = options.signal === void 0 ? consumer.signal : AbortSignal.any([options.signal, consumer.signal]);
|
|
676
|
+
const watchdog = __addDisposableResource(env_1, idleWatchdog(upstream, connection.streamIdleTimeoutMs, STREAM_IDLE_TIMEOUT_CODE), false);
|
|
677
|
+
const iterator = this.request(options, watchdog.signal, connection, apiKey, userId, () => {
|
|
678
|
+
watchdog.pulse();
|
|
679
|
+
})[Symbol.asyncIterator]();
|
|
680
|
+
let exhausted = false;
|
|
681
|
+
try {
|
|
682
|
+
while (true) {
|
|
683
|
+
const result = await watchdog.next(iterator);
|
|
684
|
+
if (result.done) {
|
|
685
|
+
exhausted = true;
|
|
686
|
+
return;
|
|
687
|
+
}
|
|
688
|
+
yield result.value;
|
|
689
|
+
}
|
|
690
|
+
} catch (error) {
|
|
691
|
+
if (timeoutOf(watchdog.signal, STREAM_IDLE_TIMEOUT_CODE) !== void 0) throw new LlmError(`SiliconFlow stream idle timeout after ${connection.streamIdleTimeoutMs}ms`, "TIMEOUT", { cause: error });
|
|
692
|
+
if (options.signal?.aborted) throw new LlmError("SiliconFlow request aborted by caller", "ABORTED", { cause: error });
|
|
693
|
+
if (error instanceof LlmError) throw error;
|
|
694
|
+
throw new LlmError(`SiliconFlow API stream from ${connection.baseURL} failed`, "TRANSPORT", { cause: error });
|
|
695
|
+
} finally {
|
|
696
|
+
consumer.abort("SiliconFlow stream consumer stopped");
|
|
697
|
+
if (!exhausted && iterator.return !== void 0) try {
|
|
698
|
+
await iterator.return();
|
|
699
|
+
} catch (_abortedTransportTeardown) {}
|
|
700
|
+
}
|
|
701
|
+
} catch (e_1) {
|
|
702
|
+
env_1.error = e_1;
|
|
703
|
+
env_1.hasError = true;
|
|
704
|
+
} finally {
|
|
705
|
+
__disposeResources(env_1);
|
|
706
|
+
}
|
|
707
|
+
}
|
|
708
|
+
async *request(options, signal, connection, apiKey, userId, onComment) {
|
|
709
|
+
const body = serializeRequest(options);
|
|
710
|
+
const payload = JSON.stringify(body);
|
|
711
|
+
const headers = {
|
|
712
|
+
"authorization": `Bearer ${apiKey}`,
|
|
713
|
+
"content-type": "application/json",
|
|
714
|
+
"accept": "text/event-stream",
|
|
715
|
+
...attributionHeaders(),
|
|
716
|
+
"x-siliconflow-harness-user-id": String(userId),
|
|
717
|
+
...options.sessionId !== void 0 ? { "x-siliconflow-harness-session-id": String(options.sessionId) } : {},
|
|
718
|
+
...options.purpose === "compaction" ? { "x-siliconflow-harness-compact": "1" } : {}
|
|
719
|
+
};
|
|
720
|
+
let response;
|
|
721
|
+
try {
|
|
722
|
+
response = await fetch(`${connection.baseURL}/chat/completions`, {
|
|
723
|
+
method: "POST",
|
|
724
|
+
headers,
|
|
725
|
+
body: payload,
|
|
726
|
+
signal
|
|
727
|
+
});
|
|
728
|
+
} catch (error) {
|
|
729
|
+
if (signal.aborted) throw error;
|
|
730
|
+
throw new LlmError(`SiliconFlow API request to ${connection.baseURL} failed`, "TRANSPORT", { cause: error });
|
|
731
|
+
}
|
|
732
|
+
if (!response.ok) {
|
|
733
|
+
let message = `SiliconFlow API error (HTTP ${response.status})`;
|
|
734
|
+
let providerError;
|
|
735
|
+
try {
|
|
736
|
+
providerError = (await response.json()).error;
|
|
737
|
+
if (providerError?.message) message = providerError.message;
|
|
738
|
+
} catch {}
|
|
739
|
+
const delay = providerRetryAfterMs(response.headers.get("retry-after"));
|
|
740
|
+
const id = requestId(response.headers);
|
|
741
|
+
throw new LlmError(message, httpErrorCode(response.status, providerError), {
|
|
742
|
+
status: response.status,
|
|
743
|
+
...delay === void 0 ? {} : { providerRetryAfterMs: delay },
|
|
744
|
+
...id === void 0 ? {} : { requestId: id }
|
|
745
|
+
});
|
|
746
|
+
}
|
|
747
|
+
if (!response.body) throw new LlmError("SiliconFlow API returned no response body", "EMPTY_RESPONSE");
|
|
748
|
+
yield* translate(parseSse(response.body, onComment));
|
|
749
|
+
}
|
|
750
|
+
};
|
|
751
|
+
//#endregion
|
|
752
|
+
//#region lib/types/index.js
|
|
753
|
+
/**
|
|
754
|
+
* Register a {@link SiliconFlowAdapter} for the `siliconflow` provider route
|
|
755
|
+
* on `ctx.llm`, with connection facts resolved per request instead of frozen
|
|
756
|
+
* at load: the plugin layers its `cordis.yml` entry config under the optional
|
|
757
|
+
* `llm-siliconflow` user-settings section (`ctx.settings`) and resolves the
|
|
758
|
+
* API key through the optional credential seam (`ctx.credentials`), so a
|
|
759
|
+
* changed base URL, catalog, or key reaches the very next request without
|
|
760
|
+
* restarting anything, while an in-flight stream keeps the facts it started
|
|
761
|
+
* with. The one registration-captured fact — the retry policy — re-registers
|
|
762
|
+
* the route in place when it changes.
|
|
763
|
+
* @module @siliconflow-official/dsh-llm-siliconflow
|
|
764
|
+
*/
|
|
765
|
+
const name = "llm-siliconflow";
|
|
766
|
+
const inject = ["llm"];
|
|
767
|
+
const NS = settingsNamespace("llm-siliconflow");
|
|
768
|
+
/** Credential reference this plugin reads by default, also used by the setup CLI. */
|
|
769
|
+
const DEFAULT_API_KEY_ENV = "SILICONFLOW_API_KEY";
|
|
770
|
+
/** The single provider route this plugin owns. */
|
|
771
|
+
const PROVIDER = "siliconflow";
|
|
772
|
+
/** Fallback advisory catalog: six widely hosted chat models, also the setup CLI's discovery fallback. */
|
|
773
|
+
const DEFAULT_MODELS = [
|
|
774
|
+
{
|
|
775
|
+
id: "zai-org/GLM-5.2",
|
|
776
|
+
contextWindow: 1e6
|
|
777
|
+
},
|
|
778
|
+
{
|
|
779
|
+
id: "moonshotai/Kimi-K2.7-Code",
|
|
780
|
+
contextWindow: 256e3
|
|
781
|
+
},
|
|
782
|
+
{
|
|
783
|
+
id: "deepseek-ai/DeepSeek-V4-Pro",
|
|
784
|
+
contextWindow: 1e6
|
|
785
|
+
},
|
|
786
|
+
{
|
|
787
|
+
id: "deepseek-ai/DeepSeek-V4-Flash",
|
|
788
|
+
contextWindow: 1e6
|
|
789
|
+
},
|
|
790
|
+
{
|
|
791
|
+
id: "Pro/moonshotai/Kimi-K2.6",
|
|
792
|
+
contextWindow: 256e3
|
|
793
|
+
},
|
|
794
|
+
{
|
|
795
|
+
id: "Qwen/Qwen3.5-397B-A17B",
|
|
796
|
+
contextWindow: 256e3
|
|
797
|
+
}
|
|
798
|
+
];
|
|
799
|
+
const catalogModel = z.object({
|
|
800
|
+
id: z.string().required(),
|
|
801
|
+
name: z.string(),
|
|
802
|
+
description: z.string(),
|
|
803
|
+
contextWindow: z.number().step(1).min(1),
|
|
804
|
+
maxTokens: z.number().step(1).min(1)
|
|
805
|
+
});
|
|
806
|
+
const Config = z.object({
|
|
807
|
+
apiKeyEnv: z.string().role("credential-ref").default(DEFAULT_API_KEY_ENV),
|
|
808
|
+
baseURL: z.string(),
|
|
809
|
+
maxTokens: z.number().step(1).min(1).max(Number.MAX_SAFE_INTEGER).default(DEFAULT_MAX_TOKENS),
|
|
810
|
+
defaultContextWindow: z.number().step(1).min(1).default(DEFAULT_CONTEXT_WINDOW),
|
|
811
|
+
models: z.array(catalogModel).default(DEFAULT_MODELS),
|
|
812
|
+
streamIdleTimeoutMs: z.number().min(Number.MIN_VALUE).max(MAX_TIMER_DELAY_MS).default(DEFAULT_STREAM_IDLE_TIMEOUT_MS),
|
|
813
|
+
retryPolicy: RetryPolicySchema
|
|
814
|
+
});
|
|
815
|
+
/** Public API default; the internal endpoint comes from $SILICONFLOW_BASE_URL. */
|
|
816
|
+
const PUBLIC_BASE_URL = "https://api.siliconflow.cn/v1";
|
|
817
|
+
/** Environment variable naming this provider's endpoint, honored only from trusted layers. */
|
|
818
|
+
const BASE_URL_ENV = "SILICONFLOW_BASE_URL";
|
|
819
|
+
/** Resolve, validate, and detach the advisory model catalog. */
|
|
820
|
+
function resolveModels(models) {
|
|
821
|
+
const seen = /* @__PURE__ */ new Set();
|
|
822
|
+
return (models ?? DEFAULT_MODELS).map((model) => {
|
|
823
|
+
if (model.id.length === 0) throw new Error("llm-siliconflow: catalog model ids must be non-empty");
|
|
824
|
+
if (model.name !== void 0 && model.name.length === 0) throw new Error(`llm-siliconflow: catalog model "${model.id}" has an empty name`);
|
|
825
|
+
if (model.contextWindow !== void 0 && (!Number.isInteger(model.contextWindow) || model.contextWindow <= 0)) throw new Error(`llm-siliconflow: catalog model "${model.id}" contextWindow must be a positive integer`);
|
|
826
|
+
if (model.maxTokens !== void 0 && (!Number.isInteger(model.maxTokens) || model.maxTokens <= 0)) throw new Error(`llm-siliconflow: catalog model "${model.id}" maxTokens must be a positive integer`);
|
|
827
|
+
if (seen.has(model.id)) throw new Error(`llm-siliconflow: duplicate catalog model "${model.id}"`);
|
|
828
|
+
seen.add(model.id);
|
|
829
|
+
return {
|
|
830
|
+
id: model.id,
|
|
831
|
+
...model.name === void 0 ? {} : { name: model.name },
|
|
832
|
+
...model.description === void 0 ? {} : { description: model.description },
|
|
833
|
+
...model.contextWindow === void 0 ? {} : { contextWindow: model.contextWindow },
|
|
834
|
+
...model.maxTokens === void 0 ? {} : { maxTokens: model.maxTokens }
|
|
835
|
+
};
|
|
836
|
+
});
|
|
837
|
+
}
|
|
838
|
+
/**
|
|
839
|
+
* The one explicit resolve step from raw config to validated connection
|
|
840
|
+
* facts. Programmatic construction may bypass Schemastery normalization, so
|
|
841
|
+
* every default and bound is re-judged here — for the composition entry at
|
|
842
|
+
* load (fail loud) and for each settings snapshot at its first use.
|
|
843
|
+
* @param config - raw plugin config or resolved settings snapshot.
|
|
844
|
+
* @param environment - this run's environment layers, or `undefined` outside
|
|
845
|
+
* the product CLI. Every layer may supply an endpoint: the product trusts the
|
|
846
|
+
* project it is launched in, so a checkout can point its own agent at the
|
|
847
|
+
* gateway that checkout is meant to use.
|
|
848
|
+
* @returns validated connection facts plus the credential reference.
|
|
849
|
+
*/
|
|
850
|
+
function resolveAdapterOptions(config, environment) {
|
|
851
|
+
if (config.defaultContextWindow !== void 0 && (!Number.isInteger(config.defaultContextWindow) || config.defaultContextWindow <= 0)) throw new Error("llm-siliconflow: defaultContextWindow must be a positive integer");
|
|
852
|
+
if (config.maxTokens !== void 0 && (!Number.isSafeInteger(config.maxTokens) || config.maxTokens <= 0)) throw new Error("llm-siliconflow: maxTokens must be a positive safe integer");
|
|
853
|
+
const streamIdleTimeoutMs = config.streamIdleTimeoutMs ?? 3e5;
|
|
854
|
+
if (!Number.isFinite(streamIdleTimeoutMs) || streamIdleTimeoutMs <= 0 || streamIdleTimeoutMs > MAX_TIMER_DELAY_MS) throw new Error(`llm-siliconflow: streamIdleTimeoutMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`);
|
|
855
|
+
return {
|
|
856
|
+
apiKeyEnv: credentialRef(config.apiKeyEnv ?? "SILICONFLOW_API_KEY"),
|
|
857
|
+
baseURL: config.baseURL ?? environment?.get(BASE_URL_ENV)?.value ?? "https://api.siliconflow.cn/v1",
|
|
858
|
+
maxTokens: config.maxTokens ?? 8192,
|
|
859
|
+
defaultContextWindow: config.defaultContextWindow ?? 32768,
|
|
860
|
+
models: resolveModels(config.models),
|
|
861
|
+
streamIdleTimeoutMs,
|
|
862
|
+
retryPolicy: resolveRetryPolicy(config.retryPolicy, "llm-siliconflow: retryPolicy")
|
|
863
|
+
};
|
|
864
|
+
}
|
|
865
|
+
function apply(ctx, config) {
|
|
866
|
+
let current = () => config;
|
|
867
|
+
let lastRaw;
|
|
868
|
+
let lastGood;
|
|
869
|
+
const options = () => {
|
|
870
|
+
const raw = current();
|
|
871
|
+
if (raw === lastRaw && lastGood !== void 0) return lastGood;
|
|
872
|
+
try {
|
|
873
|
+
const next = resolveAdapterOptions(raw, launchEnvironmentOf(ctx));
|
|
874
|
+
lastRaw = raw;
|
|
875
|
+
lastGood = next;
|
|
876
|
+
return next;
|
|
877
|
+
} catch (error) {
|
|
878
|
+
if (lastGood === void 0) throw error;
|
|
879
|
+
lastRaw = raw;
|
|
880
|
+
ctx.logger.error("llm-siliconflow: keeping the last good configuration after an invalid settings section");
|
|
881
|
+
ctx.logger.error(error);
|
|
882
|
+
return lastGood;
|
|
883
|
+
}
|
|
884
|
+
};
|
|
885
|
+
options();
|
|
886
|
+
const resolveApiKey = async (connection) => {
|
|
887
|
+
const ref = connection.apiKeyEnv;
|
|
888
|
+
const credentials = ctx.get("credentials");
|
|
889
|
+
if (credentials !== void 0) {
|
|
890
|
+
const hit = await credentials.resolve(ref);
|
|
891
|
+
if (hit !== void 0) return assertUsableApiKey(hit.value, "llm-siliconflow", ref);
|
|
892
|
+
} else {
|
|
893
|
+
const ambient = launchEnvironmentOf(ctx).get(ref);
|
|
894
|
+
if (ambient !== void 0 && ambient.value.length > 0) return assertUsableApiKey(ambient.value, "llm-siliconflow", ref);
|
|
895
|
+
}
|
|
896
|
+
throw new LlmError(`llm-siliconflow: no API key for provider route "${PROVIDER}"; store ${ref} through the credentials service (the web Models page writes it), or export ${ref} in the launching environment`, "MISSING_CREDENTIAL");
|
|
897
|
+
};
|
|
898
|
+
let userId;
|
|
899
|
+
const resolveUserId = () => userId ??= getOrCreateAnonymousUserId();
|
|
900
|
+
const storedApiKey = async () => {
|
|
901
|
+
const ref = options().apiKeyEnv;
|
|
902
|
+
const credentials = ctx.get("credentials");
|
|
903
|
+
if (credentials !== void 0) return (await credentials.resolve(ref))?.value;
|
|
904
|
+
const ambient = launchEnvironmentOf(ctx).get(ref);
|
|
905
|
+
return ambient !== void 0 && ambient.value.length > 0 ? ambient.value : void 0;
|
|
906
|
+
};
|
|
907
|
+
const adapter = new SiliconFlowAdapter({
|
|
908
|
+
options,
|
|
909
|
+
resolveApiKey,
|
|
910
|
+
resolveUserId
|
|
911
|
+
});
|
|
912
|
+
ctx.llm.registerConfigurableProviders([{
|
|
913
|
+
provider: PROVIDER,
|
|
914
|
+
displayName: "SiliconFlow",
|
|
915
|
+
settingsNs: NS,
|
|
916
|
+
settingsPath: []
|
|
917
|
+
}]);
|
|
918
|
+
ctx.llm.registerModelDiscovery(NS, async (request) => {
|
|
919
|
+
return discoverChatModels(request.baseURL ?? options().baseURL, request.apiKey ?? await storedApiKey(), request.signal);
|
|
920
|
+
});
|
|
921
|
+
const registration = ctx.llm.registerAdapter([PROVIDER], adapter);
|
|
922
|
+
let registeredPolicy = options().retryPolicy;
|
|
923
|
+
const ensureRegistrationFacts = () => {
|
|
924
|
+
const policy = options().retryPolicy;
|
|
925
|
+
if (deepEqualJson(policy, registeredPolicy)) return;
|
|
926
|
+
registration.replace([PROVIDER]);
|
|
927
|
+
registeredPolicy = policy;
|
|
928
|
+
};
|
|
929
|
+
installSettingsSection(ctx, NS, Config, config, {
|
|
930
|
+
setSource: (source) => {
|
|
931
|
+
current = source;
|
|
932
|
+
},
|
|
933
|
+
onChange: ensureRegistrationFacts
|
|
934
|
+
});
|
|
935
|
+
}
|
|
936
|
+
//#endregion
|
|
937
|
+
export { readListing as _, PUBLIC_BASE_URL as a, name as c, DEFAULT_MAX_TOKENS as d, DEFAULT_STREAM_IDLE_TIMEOUT_MS as f, listingUrl as g, discoverChatModels as h, PROVIDER as i, resolveAdapterOptions as l, SiliconFlowAdapter as m, DEFAULT_API_KEY_ENV as n, apply as o, DISCOVERY_TTL_MS as p, DEFAULT_MODELS as r, inject as s, Config as t, DEFAULT_CONTEXT_WINDOW as u };
|