@siliconflow-official/dsh-llm-siliconflow 0.1.0-rc.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,937 @@
1
+ import z from "@deepseek-ai/schemastery";
2
+ import { CONTEXT_WINDOW_EXCEEDED_CODE, CallId, EMPTY_RESPONSE_CODE, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, RetryPolicySchema, assertUsableApiKey, attributionHeaders, contentHasImage, isContextWindowExceededError, isQuotaExceededError, resolveRetryPolicy } from "@deepseek-ai/dsh-llm";
3
+ import { credentialRef } from "@deepseek-ai/dsh-credentials";
4
+ import { launchEnvironmentOf } from "@deepseek-ai/dsh-launch-environment";
5
+ import { deepEqualJson, installSettingsSection, settingsNamespace } from "@deepseek-ai/dsh-settings";
6
+ import { MAX_TIMER_DELAY_MS, idleWatchdog, timeoutOf } from "@deepseek-ai/dsh-timeout";
7
+ import { getOrCreateAnonymousUserId } from "@deepseek-ai/dsh-anonymous-user-id";
8
+ import { EventSourceParserStream } from "eventsource-parser/stream";
9
+ //#region lib/types/discovery.js
10
+ /**
11
+ * Interrogate the SiliconFlow (OpenAI-compatible) `GET /models` listing for
12
+ * the chat models an endpoint serves, filtered with `sub_type=chat` and kept
13
+ * in the endpoint's own order — SiliconFlow's listing is already ordered by
14
+ * its own preference, so the adapter must not re-sort it.
15
+ *
16
+ * This module is transport-only: it takes the endpoint and bearer token for
17
+ * one interrogation and returns the entries it read. The registering plugin
18
+ * owns credential policy, and the adapter owns caching and the fallback to its
19
+ * configured catalog.
20
+ *
21
+ * @module dsh-llm-siliconflow/discovery
22
+ */
23
+ /**
24
+ * Endpoint replies larger than this are refused. The bound holds on the bytes
25
+ * actually read, not the length the server claims, so a streaming or
26
+ * under-declaring reply cannot exhaust memory before it is turned away.
27
+ */
28
+ const MAX_RESPONSE_BYTES = 4194304;
29
+ /** A positive integer field, or `undefined` when absent or unusable. */
30
+ function capacity(...candidates) {
31
+ for (const candidate of candidates) if (typeof candidate === "number" && Number.isInteger(candidate) && candidate > 0) return candidate;
32
+ }
33
+ /** A non-empty string field, or `undefined`. */
34
+ function label(...candidates) {
35
+ for (const candidate of candidates) if (typeof candidate === "string" && candidate.length > 0) return candidate;
36
+ }
37
+ /**
38
+ * Join the endpoint base with the chat-filtered listing path. The base is a
39
+ * prefix, not a URL to resolve against, so a gateway path such as
40
+ * `https://gateway.example/openai/v1` keeps its segments.
41
+ * @param baseURL - the chat-completions base.
42
+ * @returns the `GET /models?sub_type=chat` URL.
43
+ */
44
+ function listingUrl(baseURL) {
45
+ return `${baseURL.replace(/\/+$/, "")}/models?sub_type=chat`;
46
+ }
47
+ /**
48
+ * Read a reply body, refusing one that outgrows {@link MAX_RESPONSE_BYTES}.
49
+ * @param response - the settled listing response.
50
+ * @param url - the endpoint, for the oversize diagnostic.
51
+ * @returns the decoded body text.
52
+ */
53
+ async function readBounded(response, url) {
54
+ const oversized = () => new LlmError(`${url} answered with more than ${MAX_RESPONSE_BYTES} bytes`, "DISCOVERY_FAILED");
55
+ const declared = Number(response.headers.get("content-length") ?? NaN);
56
+ if (Number.isFinite(declared) && declared > MAX_RESPONSE_BYTES) {
57
+ await response.body?.cancel();
58
+ throw oversized();
59
+ }
60
+ /* v8 ignore next -- fetch always exposes a body stream on a 2xx Response; the null guard is defensive. */
61
+ if (response.body === null) return "";
62
+ const reader = response.body.getReader();
63
+ const chunks = [];
64
+ let total = 0;
65
+ try {
66
+ for (;;) {
67
+ const { done, value } = await reader.read();
68
+ if (done) break;
69
+ total += value.byteLength;
70
+ if (total > MAX_RESPONSE_BYTES) throw oversized();
71
+ chunks.push(value);
72
+ }
73
+ } finally {
74
+ /* v8 ignore next 4 -- cancel() after a completed or abandoned read settles without rejecting; unobserved best-effort cleanup. */
75
+ await reader.cancel().catch(() => {});
76
+ }
77
+ const body = new Uint8Array(total);
78
+ let offset = 0;
79
+ for (const chunk of chunks) {
80
+ body.set(chunk, offset);
81
+ offset += chunk.byteLength;
82
+ }
83
+ return new TextDecoder().decode(body);
84
+ }
85
+ /**
86
+ * Map one listing reply into entries, preserving endpoint order. A row without
87
+ * a usable id is skipped rather than failing the whole interrogation: a single
88
+ * malformed row should not hide the rest of a working endpoint's catalog.
89
+ * @param body - the parsed reply body.
90
+ * @returns the entries in arrival order.
91
+ */
92
+ function readListing(body) {
93
+ const data = body?.data;
94
+ if (!Array.isArray(data)) throw new LlmError("the endpoint's model listing has no \"data\" array", "DISCOVERY_FAILED");
95
+ const entries = [];
96
+ for (const raw of data) {
97
+ const entry = raw;
98
+ const id = label(entry?.id);
99
+ if (id === void 0) continue;
100
+ const name = label(entry?.name, entry?.display_name);
101
+ const contextWindow = capacity(entry?.context_window, entry?.context_length);
102
+ const maxTokens = capacity(entry?.max_output_tokens, entry?.max_tokens);
103
+ entries.push({
104
+ id,
105
+ ...name === void 0 ? {} : { name },
106
+ ...contextWindow === void 0 ? {} : { contextWindow },
107
+ ...maxTokens === void 0 ? {} : { maxTokens }
108
+ });
109
+ }
110
+ return entries;
111
+ }
112
+ /**
113
+ * Interrogate one endpoint for the chat models it advertises.
114
+ * @param baseURL - the chat-completions base; `/models?sub_type=chat` is appended.
115
+ * @param apiKey - bearer token, or `undefined` to probe unauthenticated.
116
+ * @param signal - caller cancellation; the fetch and body read honor it.
117
+ * @returns the advertised models in endpoint order.
118
+ * @throws LlmError when the endpoint is unreachable, refuses the request, or
119
+ * the reply is not a readable listing.
120
+ */
121
+ async function discoverChatModels(baseURL, apiKey, signal) {
122
+ const url = listingUrl(baseURL);
123
+ let response;
124
+ try {
125
+ response = await fetch(url, {
126
+ method: "GET",
127
+ headers: {
128
+ accept: "application/json",
129
+ ...apiKey === void 0 ? {} : { authorization: `Bearer ${apiKey}` },
130
+ ...attributionHeaders()
131
+ },
132
+ ...signal === void 0 ? {} : { signal }
133
+ });
134
+ } catch (error) {
135
+ if (signal?.aborted) throw new LlmError("model discovery aborted by caller", "ABORTED", { cause: error });
136
+ throw new LlmError(`could not reach ${url}`, "DISCOVERY_FAILED", { cause: error });
137
+ }
138
+ if (!response.ok) throw new LlmError(`${url} answered ${response.status}`, "DISCOVERY_FAILED");
139
+ let text;
140
+ try {
141
+ text = await readBounded(response, url);
142
+ } catch (error) {
143
+ if (signal?.aborted) throw new LlmError("model discovery aborted by caller", "ABORTED", { cause: error });
144
+ throw error;
145
+ }
146
+ let body;
147
+ try {
148
+ body = JSON.parse(text);
149
+ } catch (error) {
150
+ throw new LlmError(`${url} did not answer with JSON`, "DISCOVERY_FAILED", { cause: error });
151
+ }
152
+ return readListing(body);
153
+ }
154
+ //#endregion
155
+ //#region lib/types/serialize.js
156
+ /**
157
+ * Serialize harness messages into SiliconFlow chat completions. User text is
158
+ * joined; assistant text becomes `content`, tool calls become `tool_calls`,
159
+ * and tool results become separate tool messages. Assistant reasoning is
160
+ * replayed as `reasoning_content` only on tool-call turns, as hosted reasoning
161
+ * models (DeepSeek-R1 and siblings) require. Core image blocks are rejected
162
+ * explicitly because this wire route is text-only; unknown declaration-merged
163
+ * block types retain the adapter's documented extension fallback.
164
+ * @module dsh-llm-siliconflow/serialize
165
+ */
166
+ /** Join the text blocks of a message (used for user/tool-result content). */
167
+ function flattenText(blocks) {
168
+ return blocks.filter((block) => block.type === "text").map((block) => block.text).join("");
169
+ }
170
+ /** Reject core image content before any text-flattening path can silently erase it. */
171
+ function assertTextOnly(blocks) {
172
+ if (contentHasImage(blocks)) throw new LlmError("The SiliconFlow chat-completions adapter does not support image content.", "UNSUPPORTED_CONTENT");
173
+ }
174
+ /** Serialize one assistant message (text + reasoning + tool calls). */
175
+ function serializeAssistant(message) {
176
+ const text = flattenText(message.content);
177
+ const reasoning = message.content.filter((block) => block.type === "reasoning").map((block) => block.text).join("");
178
+ const toolCalls = message.content.filter((block) => block.type === "tool-call").map((block) => ({
179
+ id: block.id,
180
+ type: "function",
181
+ function: {
182
+ name: block.name,
183
+ arguments: block.arguments
184
+ }
185
+ }));
186
+ return {
187
+ role: "assistant",
188
+ content: text,
189
+ ...toolCalls.length > 0 && reasoning.length > 0 ? { reasoning_content: reasoning } : {},
190
+ ...toolCalls.length > 0 ? { tool_calls: toolCalls } : {}
191
+ };
192
+ }
193
+ /**
194
+ * Serialize the conversation. `tool-result` blocks become standalone
195
+ * `{role: 'tool'}` messages; the harness puts each tool result in its own
196
+ * user-role message, so a mixed user message contributes its text first and
197
+ * its tool results as separate wire messages after.
198
+ * @param messages - the harness conversation, in order.
199
+ * @returns the wire messages; order preserved, each tool result expanded into its own entry.
200
+ */
201
+ function serializeMessages(messages) {
202
+ const wire = [];
203
+ for (const message of messages) {
204
+ assertTextOnly(message.content);
205
+ if (message.role === "system") {
206
+ wire.push({
207
+ role: "system",
208
+ content: flattenText(message.content)
209
+ });
210
+ continue;
211
+ }
212
+ if (message.role === "assistant") {
213
+ wire.push(serializeAssistant(message));
214
+ continue;
215
+ }
216
+ const toolResults = message.content.filter((block) => block.type === "tool-result");
217
+ const text = flattenText(message.content);
218
+ if (text.length > 0 || toolResults.length === 0) wire.push({
219
+ role: "user",
220
+ content: text
221
+ });
222
+ for (const result of toolResults) wire.push({
223
+ role: "tool",
224
+ tool_call_id: result.toolCallId,
225
+ content: flattenText(result.content) || "(no output)"
226
+ });
227
+ }
228
+ return wire;
229
+ }
230
+ /**
231
+ * Build the full wire request. Always streaming (`stream: true`, usage
232
+ * reporting on); optional fields are omitted rather than sent as null, so
233
+ * provider defaults apply.
234
+ * @param options - the harness request (model, history, system, tools, sampling).
235
+ * @returns the chat-completions request body.
236
+ */
237
+ function serializeRequest(options) {
238
+ const messages = [];
239
+ if (options.system !== void 0) messages.push({
240
+ role: "system",
241
+ content: options.system
242
+ });
243
+ messages.push(...serializeMessages(options.messages));
244
+ const tools = options.tools?.map((tool) => ({
245
+ type: "function",
246
+ function: {
247
+ name: tool.name,
248
+ description: tool.description,
249
+ parameters: tool.parameters
250
+ }
251
+ }));
252
+ return {
253
+ model: options.model,
254
+ messages,
255
+ stream: true,
256
+ stream_options: { include_usage: true },
257
+ ...tools !== void 0 && tools.length > 0 ? { tools } : {},
258
+ ...options.temperature !== void 0 ? { temperature: options.temperature } : {},
259
+ ...options.maxTokens === void 0 ? {} : { max_tokens: options.maxTokens },
260
+ ...options.stop !== void 0 ? { stop: options.stop } : {}
261
+ };
262
+ }
263
+ /**
264
+ * Parse an SSE byte stream into data payloads. Yields `[DONE]` as the final
265
+ * value and returns; throws `LlmError('STREAM_CLOSED')` when the stream ends
266
+ * without it (truncated response — the model call cannot be trusted).
267
+ * @param stream - raw SSE bytes; reads may split anywhere, including mid-UTF-8 sequence.
268
+ * @param onComment - optional transport-activity callback; comments never enter the yielded payload stream.
269
+ * @returns each event's data payload in arrival order, the `[DONE]` sentinel last.
270
+ */
271
+ async function* parseSse(stream, onComment) {
272
+ const events = stream.pipeThrough(new TextDecoderStream()).pipeThrough(new EventSourceParserStream({ onComment }));
273
+ for await (const { data } of events) {
274
+ yield data;
275
+ if (data === "[DONE]") return;
276
+ }
277
+ throw new LlmError("SSE stream ended without [DONE]", "STREAM_CLOSED");
278
+ }
279
+ //#endregion
280
+ //#region lib/types/translate.js
281
+ /**
282
+ * Translate SiliconFlow SSE payloads with one stateful harness block per
283
+ * content, reasoning, or tool-call index. An empty initial reasoning delta
284
+ * does not open a block. Finish reason and the latest usage are deferred until
285
+ * `[DONE]`, covering both finish-attached and trailing usage-only shapes while
286
+ * ensuring no chunk follows `finish`.
287
+ *
288
+ * Translate SiliconFlow wire chunks into the harness `StreamChunk` protocol.
289
+ * @module dsh-llm-siliconflow/translate
290
+ */
291
+ /**
292
+ * Map the wire finish_reason vocabulary to the harness FinishReason.
293
+ * @param reason - the wire `finish_reason` string.
294
+ * @returns the mapped reason; unrecognized values (content_filter, …) become `{kind: 'error'}` with the uppercased value as `code`.
295
+ */
296
+ function mapFinishReason(reason) {
297
+ switch (reason) {
298
+ case "stop": return { kind: "stop" };
299
+ case "tool_calls": return { kind: "tool-calls" };
300
+ case "length": return { kind: "max-tokens" };
301
+ default: return {
302
+ kind: "error",
303
+ failure: {
304
+ message: `model stopped: ${reason}`,
305
+ code: reason.toUpperCase()
306
+ }
307
+ };
308
+ }
309
+ }
310
+ /**
311
+ * Map wire usage fields. SiliconFlow's `prompt_tokens` INCLUDES cache hits
312
+ * (`prompt_tokens = prompt_cache_hit_tokens + prompt_cache_miss_tokens`); the
313
+ * harness TokenUsage convention is DISJOINT counts, so cache reads are
314
+ * subtracted out of `inputTokens`.
315
+ * @param usage - wire usage from the finish chunk or the trailing usage-only chunk.
316
+ * @returns disjoint harness counts; cache/reasoning fields present only when the wire reported them.
317
+ */
318
+ function mapUsage(usage) {
319
+ const cacheRead = usage.prompt_tokens_details?.cached_tokens ?? usage.prompt_cache_hit_tokens;
320
+ const reasoning = usage.completion_tokens_details?.reasoning_tokens;
321
+ return {
322
+ inputTokens: usage.prompt_tokens - (cacheRead ?? 0),
323
+ outputTokens: usage.completion_tokens,
324
+ ...cacheRead !== void 0 ? { cacheReadTokens: cacheRead } : {},
325
+ ...reasoning !== void 0 ? { reasoningTokens: reasoning } : {}
326
+ };
327
+ }
328
+ /** Assemble the final ContentBlock for one open block. */
329
+ function closeBlock(block) {
330
+ switch (block.kind) {
331
+ case "text": return {
332
+ type: "text",
333
+ text: block.text
334
+ };
335
+ case "reasoning": return {
336
+ type: "reasoning",
337
+ text: block.text
338
+ };
339
+ case "tool-call": return {
340
+ type: "tool-call",
341
+ id: CallId(block.callId ?? ""),
342
+ name: block.name ?? "",
343
+ arguments: block.text
344
+ };
345
+ }
346
+ }
347
+ /**
348
+ * Consume SSE data payloads (ending with `[DONE]`) and yield StreamChunks.
349
+ * Malformed JSON payloads abort the stream with `MALFORMED_RESPONSE`.
350
+ * @param payloads - SSE data payloads from {@link parseSse}, `[DONE]`-terminated.
351
+ * @returns deltas as they arrive; `block-end`s, `usage`, and `finish` are all deferred to the `[DONE]` sentinel.
352
+ * A `stop` (or absent) finish with no opened blocks is a degenerate provider completion and maps to an
353
+ * `EMPTY_RESPONSE` error finish instead of a successful empty message.
354
+ */
355
+ async function* translate(payloads) {
356
+ let nextIndex = 0;
357
+ let textBlock;
358
+ let reasoningBlock;
359
+ const toolBlocks = /* @__PURE__ */ new Map();
360
+ const order = [];
361
+ let pendingFinish;
362
+ let pendingUsage;
363
+ function open(kind) {
364
+ const block = {
365
+ index: nextIndex++,
366
+ kind,
367
+ text: ""
368
+ };
369
+ order.push(block);
370
+ return block;
371
+ }
372
+ for await (const payload of payloads) {
373
+ if (payload === "[DONE]") {
374
+ for (const block of order) yield {
375
+ type: "block-end",
376
+ index: block.index,
377
+ block: closeBlock(block)
378
+ };
379
+ if (pendingUsage) yield {
380
+ type: "usage",
381
+ usage: pendingUsage
382
+ };
383
+ const reason = pendingFinish ?? { kind: "stop" };
384
+ yield {
385
+ type: "finish",
386
+ reason: reason.kind === "stop" && order.length === 0 ? {
387
+ kind: "error",
388
+ failure: {
389
+ message: "model returned a completed response with no content",
390
+ code: EMPTY_RESPONSE_CODE
391
+ }
392
+ } : reason
393
+ };
394
+ return;
395
+ }
396
+ let chunk;
397
+ try {
398
+ chunk = JSON.parse(payload);
399
+ } catch {
400
+ throw new LlmError(`malformed SSE payload: ${payload.slice(0, 120)}`, "MALFORMED_RESPONSE");
401
+ }
402
+ for (const choice of chunk.choices ?? []) {
403
+ const delta = choice.delta;
404
+ const reasoning = delta?.reasoning_content;
405
+ if (typeof reasoning === "string" && reasoning.length > 0) {
406
+ if (!reasoningBlock) {
407
+ reasoningBlock = open("reasoning");
408
+ yield {
409
+ type: "block-start",
410
+ index: reasoningBlock.index,
411
+ blockType: "reasoning"
412
+ };
413
+ }
414
+ reasoningBlock.text += reasoning;
415
+ yield {
416
+ type: "reasoning-delta",
417
+ index: reasoningBlock.index,
418
+ text: reasoning
419
+ };
420
+ }
421
+ const content = delta?.content;
422
+ if (typeof content === "string" && content.length > 0) {
423
+ if (!textBlock) {
424
+ textBlock = open("text");
425
+ yield {
426
+ type: "block-start",
427
+ index: textBlock.index,
428
+ blockType: "text"
429
+ };
430
+ }
431
+ textBlock.text += content;
432
+ yield {
433
+ type: "text-delta",
434
+ index: textBlock.index,
435
+ text: content
436
+ };
437
+ }
438
+ for (const call of delta?.tool_calls ?? []) {
439
+ let block = toolBlocks.get(call.index);
440
+ if (!block) {
441
+ block = open("tool-call");
442
+ toolBlocks.set(call.index, block);
443
+ yield {
444
+ type: "block-start",
445
+ index: block.index,
446
+ blockType: "tool-call"
447
+ };
448
+ }
449
+ if (call.id !== void 0) block.callId = call.id;
450
+ if (call.function?.name !== void 0) block.name = call.function.name;
451
+ const fragment = call.function?.arguments ?? "";
452
+ block.text += fragment;
453
+ yield {
454
+ type: "tool-call-delta",
455
+ index: block.index,
456
+ id: CallId(block.callId ?? ""),
457
+ ...block.name !== void 0 ? { name: block.name } : {},
458
+ argumentsDelta: fragment
459
+ };
460
+ }
461
+ if (typeof choice.finish_reason === "string") pendingFinish = mapFinishReason(choice.finish_reason);
462
+ }
463
+ if (chunk.usage) pendingUsage = mapUsage(chunk.usage);
464
+ }
465
+ throw new LlmError("SSE payload stream ended without [DONE]", "STREAM_CLOSED");
466
+ }
467
+ //#endregion
468
+ //#region lib/types/adapter.js
469
+ /**
470
+ * `SiliconFlowAdapter`: fetch + SSE against a SiliconFlow (OpenAI-compatible)
471
+ * chat-completions endpoint, emitting harness StreamChunks. The adapter is
472
+ * transport-only: connection facts arrive through a thunk resolved once per
473
+ * operation and the bearer token through a per-request resolver, so the
474
+ * registering plugin owns validation, layering, and credential policy.
475
+ *
476
+ * @module dsh-llm-siliconflow/adapter
477
+ */
478
+ var __addDisposableResource = function(env, value, async) {
479
+ if (value !== null && value !== void 0) {
480
+ if (typeof value !== "object" && typeof value !== "function") throw new TypeError("Object expected.");
481
+ var dispose, inner;
482
+ if (async) {
483
+ if (!Symbol.asyncDispose) throw new TypeError("Symbol.asyncDispose is not defined.");
484
+ dispose = value[Symbol.asyncDispose];
485
+ }
486
+ if (dispose === void 0) {
487
+ if (!Symbol.dispose) throw new TypeError("Symbol.dispose is not defined.");
488
+ dispose = value[Symbol.dispose];
489
+ if (async) inner = dispose;
490
+ }
491
+ if (typeof dispose !== "function") throw new TypeError("Object not disposable.");
492
+ if (inner) dispose = function() {
493
+ try {
494
+ inner.call(this);
495
+ } catch (e) {
496
+ return Promise.reject(e);
497
+ }
498
+ };
499
+ env.stack.push({
500
+ value,
501
+ dispose,
502
+ async
503
+ });
504
+ } else if (async) env.stack.push({ async: true });
505
+ return value;
506
+ };
507
+ var __disposeResources = (function(SuppressedError) {
508
+ return function(env) {
509
+ function fail(e) {
510
+ env.error = env.hasError ? new SuppressedError(e, env.error, "An error was suppressed during disposal.") : e;
511
+ env.hasError = true;
512
+ }
513
+ var r, s = 0;
514
+ function next() {
515
+ while (r = env.stack.pop()) try {
516
+ if (!r.async && s === 1) return s = 0, env.stack.push(r), Promise.resolve().then(next);
517
+ if (r.dispose) {
518
+ var result = r.dispose.call(r.value);
519
+ if (r.async) return s |= 2, Promise.resolve(result).then(next, function(e) {
520
+ fail(e);
521
+ return next();
522
+ });
523
+ } else s |= 1;
524
+ } catch (e) {
525
+ fail(e);
526
+ }
527
+ if (s === 1) return env.hasError ? Promise.reject(env.error) : Promise.resolve();
528
+ if (env.hasError) throw env.error;
529
+ }
530
+ return next();
531
+ };
532
+ })(typeof SuppressedError === "function" ? SuppressedError : function(error, suppressed, message) {
533
+ var e = new Error(message);
534
+ return e.name = "SuppressedError", e.error = error, e.suppressed = suppressed, e;
535
+ });
536
+ /** Default maximum idle interval while an adapter stream read is outstanding. */
537
+ const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 3e5;
538
+ /** Default combined request/response context capacity. */
539
+ const DEFAULT_CONTEXT_WINDOW = 32768;
540
+ /** Default per-request output-token cap. */
541
+ const DEFAULT_MAX_TOKENS = 8192;
542
+ /** How long a cached model-listing discovery stays fresh before the next `listModels` re-interrogates. */
543
+ const DISCOVERY_TTL_MS = 3e5;
544
+ const STREAM_IDLE_TIMEOUT_CODE = "LLM_STREAM_IDLE_TIMEOUT";
545
+ function modelInfo(provider, model) {
546
+ return {
547
+ provider,
548
+ id: model.id,
549
+ name: model.name ?? model.id,
550
+ ...model.description === void 0 ? {} : { description: model.description },
551
+ inputModalities: ["text"]
552
+ };
553
+ }
554
+ function providerRetryAfterMs(value) {
555
+ if (value === null) return void 0;
556
+ if (/^\d+$/.test(value)) {
557
+ const delay = Number(value) * 1e3;
558
+ return Number.isFinite(delay) && delay > 0 ? delay : void 0;
559
+ }
560
+ const delay = Date.parse(value) - Date.now();
561
+ return Number.isFinite(delay) && delay > 0 ? delay : void 0;
562
+ }
563
+ function requestId(headers) {
564
+ const value = headers.get("x-request-id");
565
+ return value === null || value.length === 0 ? void 0 : ProviderRequestId(value);
566
+ }
567
+ /**
568
+ * Map an HTTP status to a stable LlmError code.
569
+ * @param status - status of a non-2xx provider response.
570
+ * @param error - parsed provider error body, when available.
571
+ * @returns the normalized harness error code.
572
+ */
573
+ function httpErrorCode(status, error) {
574
+ if (status === 401 || status === 403) return "AUTH";
575
+ const detail = [
576
+ error?.code,
577
+ error?.type,
578
+ error?.message
579
+ ].filter(Boolean).join(" ");
580
+ if (isQuotaExceededError(detail)) return QUOTA_EXCEEDED_CODE;
581
+ if (status === 429) return "RATE_LIMIT";
582
+ if (status === 400) {
583
+ if (isContextWindowExceededError(detail)) return CONTEXT_WINDOW_EXCEEDED_CODE;
584
+ return "INVALID_REQUEST";
585
+ }
586
+ if (status >= 500) return "SERVER";
587
+ return `HTTP_${status}`;
588
+ }
589
+ /**
590
+ * A direct-fetch `LlmAdapter` for SiliconFlow's OpenAI-compatible
591
+ * chat-completions endpoint. One instance serves every model name it was
592
+ * registered under (the harness model name IS the wire model name).
593
+ *
594
+ * One stable signal reaches both initial fetch and body reads. Caller aborts
595
+ * map to `ABORTED`; the configured per-read idle watchdog maps to `TIMEOUT`.
596
+ */
597
+ var SiliconFlowAdapter = class extends LlmAdapter {
598
+ config;
599
+ /** Cached listing result, keyed by the baseURL it was read from. */
600
+ discovery;
601
+ constructor(config) {
602
+ super();
603
+ this.config = config;
604
+ }
605
+ providerInfo(provider) {
606
+ return {
607
+ id: provider,
608
+ name: "SiliconFlow"
609
+ };
610
+ }
611
+ providerRetryPolicy(_provider) {
612
+ return this.config.options().retryPolicy;
613
+ }
614
+ /** The cached listing for `connection` when it is still fresh; never re-interrogates. */
615
+ freshDiscovery(connection) {
616
+ if (this.discovery?.baseURL !== connection.baseURL) return void 0;
617
+ if (Date.now() - this.discovery.fetchedAt > 3e5) return void 0;
618
+ return this.discovery.models;
619
+ }
620
+ /**
621
+ * The models this adapter currently advertises: the live chat listing when a
622
+ * key and a reachable endpoint can supply one, else the configured catalog.
623
+ * Discovery is advisory and best-effort — a missing key or any interrogation
624
+ * failure falls back to the configured `models` rather than breaking the
625
+ * picker, because an absent catalog would hide the provider entirely.
626
+ */
627
+ async discover(connection) {
628
+ const cached = this.freshDiscovery(connection);
629
+ if (cached !== void 0) return cached;
630
+ try {
631
+ const apiKey = await this.config.resolveApiKey(connection);
632
+ const models = await discoverChatModels(connection.baseURL, apiKey);
633
+ this.discovery = {
634
+ baseURL: connection.baseURL,
635
+ models: [...models],
636
+ fetchedAt: Date.now()
637
+ };
638
+ return this.discovery.models;
639
+ } catch {
640
+ this.discovery = void 0;
641
+ return connection.models;
642
+ }
643
+ }
644
+ async listModels(provider) {
645
+ const connection = this.config.options();
646
+ return (await this.discover(connection)).map((model) => modelInfo(provider, model));
647
+ }
648
+ resolveModel(provider, model, _signal) {
649
+ const connection = this.config.options();
650
+ const configured = connection.models.find((entry) => entry.id === model);
651
+ const discovered = this.freshDiscovery(connection)?.find((entry) => entry.id === model);
652
+ const entry = configured ?? discovered;
653
+ return Promise.resolve({
654
+ ...entry === void 0 ? {
655
+ provider,
656
+ id: model,
657
+ name: model,
658
+ inputModalities: ["text"]
659
+ } : modelInfo(provider, entry),
660
+ context: { contextWindow: entry?.contextWindow ?? connection.defaultContextWindow },
661
+ defaultMaxTokens: entry?.maxTokens ?? connection.maxTokens
662
+ });
663
+ }
664
+ async *stream(options) {
665
+ const env_1 = {
666
+ stack: [],
667
+ error: void 0,
668
+ hasError: false
669
+ };
670
+ try {
671
+ const connection = this.config.options();
672
+ const apiKey = await this.config.resolveApiKey(connection);
673
+ const userId = this.config.resolveUserId();
674
+ const consumer = new AbortController();
675
+ const upstream = options.signal === void 0 ? consumer.signal : AbortSignal.any([options.signal, consumer.signal]);
676
+ const watchdog = __addDisposableResource(env_1, idleWatchdog(upstream, connection.streamIdleTimeoutMs, STREAM_IDLE_TIMEOUT_CODE), false);
677
+ const iterator = this.request(options, watchdog.signal, connection, apiKey, userId, () => {
678
+ watchdog.pulse();
679
+ })[Symbol.asyncIterator]();
680
+ let exhausted = false;
681
+ try {
682
+ while (true) {
683
+ const result = await watchdog.next(iterator);
684
+ if (result.done) {
685
+ exhausted = true;
686
+ return;
687
+ }
688
+ yield result.value;
689
+ }
690
+ } catch (error) {
691
+ if (timeoutOf(watchdog.signal, STREAM_IDLE_TIMEOUT_CODE) !== void 0) throw new LlmError(`SiliconFlow stream idle timeout after ${connection.streamIdleTimeoutMs}ms`, "TIMEOUT", { cause: error });
692
+ if (options.signal?.aborted) throw new LlmError("SiliconFlow request aborted by caller", "ABORTED", { cause: error });
693
+ if (error instanceof LlmError) throw error;
694
+ throw new LlmError(`SiliconFlow API stream from ${connection.baseURL} failed`, "TRANSPORT", { cause: error });
695
+ } finally {
696
+ consumer.abort("SiliconFlow stream consumer stopped");
697
+ if (!exhausted && iterator.return !== void 0) try {
698
+ await iterator.return();
699
+ } catch (_abortedTransportTeardown) {}
700
+ }
701
+ } catch (e_1) {
702
+ env_1.error = e_1;
703
+ env_1.hasError = true;
704
+ } finally {
705
+ __disposeResources(env_1);
706
+ }
707
+ }
708
+ async *request(options, signal, connection, apiKey, userId, onComment) {
709
+ const body = serializeRequest(options);
710
+ const payload = JSON.stringify(body);
711
+ const headers = {
712
+ "authorization": `Bearer ${apiKey}`,
713
+ "content-type": "application/json",
714
+ "accept": "text/event-stream",
715
+ ...attributionHeaders(),
716
+ "x-siliconflow-harness-user-id": String(userId),
717
+ ...options.sessionId !== void 0 ? { "x-siliconflow-harness-session-id": String(options.sessionId) } : {},
718
+ ...options.purpose === "compaction" ? { "x-siliconflow-harness-compact": "1" } : {}
719
+ };
720
+ let response;
721
+ try {
722
+ response = await fetch(`${connection.baseURL}/chat/completions`, {
723
+ method: "POST",
724
+ headers,
725
+ body: payload,
726
+ signal
727
+ });
728
+ } catch (error) {
729
+ if (signal.aborted) throw error;
730
+ throw new LlmError(`SiliconFlow API request to ${connection.baseURL} failed`, "TRANSPORT", { cause: error });
731
+ }
732
+ if (!response.ok) {
733
+ let message = `SiliconFlow API error (HTTP ${response.status})`;
734
+ let providerError;
735
+ try {
736
+ providerError = (await response.json()).error;
737
+ if (providerError?.message) message = providerError.message;
738
+ } catch {}
739
+ const delay = providerRetryAfterMs(response.headers.get("retry-after"));
740
+ const id = requestId(response.headers);
741
+ throw new LlmError(message, httpErrorCode(response.status, providerError), {
742
+ status: response.status,
743
+ ...delay === void 0 ? {} : { providerRetryAfterMs: delay },
744
+ ...id === void 0 ? {} : { requestId: id }
745
+ });
746
+ }
747
+ if (!response.body) throw new LlmError("SiliconFlow API returned no response body", "EMPTY_RESPONSE");
748
+ yield* translate(parseSse(response.body, onComment));
749
+ }
750
+ };
751
+ //#endregion
752
+ //#region lib/types/index.js
753
+ /**
754
+ * Register a {@link SiliconFlowAdapter} for the `siliconflow` provider route
755
+ * on `ctx.llm`, with connection facts resolved per request instead of frozen
756
+ * at load: the plugin layers its `cordis.yml` entry config under the optional
757
+ * `llm-siliconflow` user-settings section (`ctx.settings`) and resolves the
758
+ * API key through the optional credential seam (`ctx.credentials`), so a
759
+ * changed base URL, catalog, or key reaches the very next request without
760
+ * restarting anything, while an in-flight stream keeps the facts it started
761
+ * with. The one registration-captured fact — the retry policy — re-registers
762
+ * the route in place when it changes.
763
+ * @module @siliconflow-official/dsh-llm-siliconflow
764
+ */
765
+ const name = "llm-siliconflow";
766
+ const inject = ["llm"];
767
+ const NS = settingsNamespace("llm-siliconflow");
768
+ /** Credential reference this plugin reads by default, also used by the setup CLI. */
769
+ const DEFAULT_API_KEY_ENV = "SILICONFLOW_API_KEY";
770
+ /** The single provider route this plugin owns. */
771
+ const PROVIDER = "siliconflow";
772
+ /** Fallback advisory catalog: six widely hosted chat models, also the setup CLI's discovery fallback. */
773
+ const DEFAULT_MODELS = [
774
+ {
775
+ id: "zai-org/GLM-5.2",
776
+ contextWindow: 1e6
777
+ },
778
+ {
779
+ id: "moonshotai/Kimi-K2.7-Code",
780
+ contextWindow: 256e3
781
+ },
782
+ {
783
+ id: "deepseek-ai/DeepSeek-V4-Pro",
784
+ contextWindow: 1e6
785
+ },
786
+ {
787
+ id: "deepseek-ai/DeepSeek-V4-Flash",
788
+ contextWindow: 1e6
789
+ },
790
+ {
791
+ id: "Pro/moonshotai/Kimi-K2.6",
792
+ contextWindow: 256e3
793
+ },
794
+ {
795
+ id: "Qwen/Qwen3.5-397B-A17B",
796
+ contextWindow: 256e3
797
+ }
798
+ ];
799
+ const catalogModel = z.object({
800
+ id: z.string().required(),
801
+ name: z.string(),
802
+ description: z.string(),
803
+ contextWindow: z.number().step(1).min(1),
804
+ maxTokens: z.number().step(1).min(1)
805
+ });
806
+ const Config = z.object({
807
+ apiKeyEnv: z.string().role("credential-ref").default(DEFAULT_API_KEY_ENV),
808
+ baseURL: z.string(),
809
+ maxTokens: z.number().step(1).min(1).max(Number.MAX_SAFE_INTEGER).default(DEFAULT_MAX_TOKENS),
810
+ defaultContextWindow: z.number().step(1).min(1).default(DEFAULT_CONTEXT_WINDOW),
811
+ models: z.array(catalogModel).default(DEFAULT_MODELS),
812
+ streamIdleTimeoutMs: z.number().min(Number.MIN_VALUE).max(MAX_TIMER_DELAY_MS).default(DEFAULT_STREAM_IDLE_TIMEOUT_MS),
813
+ retryPolicy: RetryPolicySchema
814
+ });
815
+ /** Public API default; the internal endpoint comes from $SILICONFLOW_BASE_URL. */
816
+ const PUBLIC_BASE_URL = "https://api.siliconflow.cn/v1";
817
+ /** Environment variable naming this provider's endpoint, honored only from trusted layers. */
818
+ const BASE_URL_ENV = "SILICONFLOW_BASE_URL";
819
+ /** Resolve, validate, and detach the advisory model catalog. */
820
+ function resolveModels(models) {
821
+ const seen = /* @__PURE__ */ new Set();
822
+ return (models ?? DEFAULT_MODELS).map((model) => {
823
+ if (model.id.length === 0) throw new Error("llm-siliconflow: catalog model ids must be non-empty");
824
+ if (model.name !== void 0 && model.name.length === 0) throw new Error(`llm-siliconflow: catalog model "${model.id}" has an empty name`);
825
+ if (model.contextWindow !== void 0 && (!Number.isInteger(model.contextWindow) || model.contextWindow <= 0)) throw new Error(`llm-siliconflow: catalog model "${model.id}" contextWindow must be a positive integer`);
826
+ if (model.maxTokens !== void 0 && (!Number.isInteger(model.maxTokens) || model.maxTokens <= 0)) throw new Error(`llm-siliconflow: catalog model "${model.id}" maxTokens must be a positive integer`);
827
+ if (seen.has(model.id)) throw new Error(`llm-siliconflow: duplicate catalog model "${model.id}"`);
828
+ seen.add(model.id);
829
+ return {
830
+ id: model.id,
831
+ ...model.name === void 0 ? {} : { name: model.name },
832
+ ...model.description === void 0 ? {} : { description: model.description },
833
+ ...model.contextWindow === void 0 ? {} : { contextWindow: model.contextWindow },
834
+ ...model.maxTokens === void 0 ? {} : { maxTokens: model.maxTokens }
835
+ };
836
+ });
837
+ }
838
+ /**
839
+ * The one explicit resolve step from raw config to validated connection
840
+ * facts. Programmatic construction may bypass Schemastery normalization, so
841
+ * every default and bound is re-judged here — for the composition entry at
842
+ * load (fail loud) and for each settings snapshot at its first use.
843
+ * @param config - raw plugin config or resolved settings snapshot.
844
+ * @param environment - this run's environment layers, or `undefined` outside
845
+ * the product CLI. Every layer may supply an endpoint: the product trusts the
846
+ * project it is launched in, so a checkout can point its own agent at the
847
+ * gateway that checkout is meant to use.
848
+ * @returns validated connection facts plus the credential reference.
849
+ */
850
+ function resolveAdapterOptions(config, environment) {
851
+ if (config.defaultContextWindow !== void 0 && (!Number.isInteger(config.defaultContextWindow) || config.defaultContextWindow <= 0)) throw new Error("llm-siliconflow: defaultContextWindow must be a positive integer");
852
+ if (config.maxTokens !== void 0 && (!Number.isSafeInteger(config.maxTokens) || config.maxTokens <= 0)) throw new Error("llm-siliconflow: maxTokens must be a positive safe integer");
853
+ const streamIdleTimeoutMs = config.streamIdleTimeoutMs ?? 3e5;
854
+ if (!Number.isFinite(streamIdleTimeoutMs) || streamIdleTimeoutMs <= 0 || streamIdleTimeoutMs > MAX_TIMER_DELAY_MS) throw new Error(`llm-siliconflow: streamIdleTimeoutMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`);
855
+ return {
856
+ apiKeyEnv: credentialRef(config.apiKeyEnv ?? "SILICONFLOW_API_KEY"),
857
+ baseURL: config.baseURL ?? environment?.get(BASE_URL_ENV)?.value ?? "https://api.siliconflow.cn/v1",
858
+ maxTokens: config.maxTokens ?? 8192,
859
+ defaultContextWindow: config.defaultContextWindow ?? 32768,
860
+ models: resolveModels(config.models),
861
+ streamIdleTimeoutMs,
862
+ retryPolicy: resolveRetryPolicy(config.retryPolicy, "llm-siliconflow: retryPolicy")
863
+ };
864
+ }
865
+ function apply(ctx, config) {
866
+ let current = () => config;
867
+ let lastRaw;
868
+ let lastGood;
869
+ const options = () => {
870
+ const raw = current();
871
+ if (raw === lastRaw && lastGood !== void 0) return lastGood;
872
+ try {
873
+ const next = resolveAdapterOptions(raw, launchEnvironmentOf(ctx));
874
+ lastRaw = raw;
875
+ lastGood = next;
876
+ return next;
877
+ } catch (error) {
878
+ if (lastGood === void 0) throw error;
879
+ lastRaw = raw;
880
+ ctx.logger.error("llm-siliconflow: keeping the last good configuration after an invalid settings section");
881
+ ctx.logger.error(error);
882
+ return lastGood;
883
+ }
884
+ };
885
+ options();
886
+ const resolveApiKey = async (connection) => {
887
+ const ref = connection.apiKeyEnv;
888
+ const credentials = ctx.get("credentials");
889
+ if (credentials !== void 0) {
890
+ const hit = await credentials.resolve(ref);
891
+ if (hit !== void 0) return assertUsableApiKey(hit.value, "llm-siliconflow", ref);
892
+ } else {
893
+ const ambient = launchEnvironmentOf(ctx).get(ref);
894
+ if (ambient !== void 0 && ambient.value.length > 0) return assertUsableApiKey(ambient.value, "llm-siliconflow", ref);
895
+ }
896
+ throw new LlmError(`llm-siliconflow: no API key for provider route "${PROVIDER}"; store ${ref} through the credentials service (the web Models page writes it), or export ${ref} in the launching environment`, "MISSING_CREDENTIAL");
897
+ };
898
+ let userId;
899
+ const resolveUserId = () => userId ??= getOrCreateAnonymousUserId();
900
+ const storedApiKey = async () => {
901
+ const ref = options().apiKeyEnv;
902
+ const credentials = ctx.get("credentials");
903
+ if (credentials !== void 0) return (await credentials.resolve(ref))?.value;
904
+ const ambient = launchEnvironmentOf(ctx).get(ref);
905
+ return ambient !== void 0 && ambient.value.length > 0 ? ambient.value : void 0;
906
+ };
907
+ const adapter = new SiliconFlowAdapter({
908
+ options,
909
+ resolveApiKey,
910
+ resolveUserId
911
+ });
912
+ ctx.llm.registerConfigurableProviders([{
913
+ provider: PROVIDER,
914
+ displayName: "SiliconFlow",
915
+ settingsNs: NS,
916
+ settingsPath: []
917
+ }]);
918
+ ctx.llm.registerModelDiscovery(NS, async (request) => {
919
+ return discoverChatModels(request.baseURL ?? options().baseURL, request.apiKey ?? await storedApiKey(), request.signal);
920
+ });
921
+ const registration = ctx.llm.registerAdapter([PROVIDER], adapter);
922
+ let registeredPolicy = options().retryPolicy;
923
+ const ensureRegistrationFacts = () => {
924
+ const policy = options().retryPolicy;
925
+ if (deepEqualJson(policy, registeredPolicy)) return;
926
+ registration.replace([PROVIDER]);
927
+ registeredPolicy = policy;
928
+ };
929
+ installSettingsSection(ctx, NS, Config, config, {
930
+ setSource: (source) => {
931
+ current = source;
932
+ },
933
+ onChange: ensureRegistrationFacts
934
+ });
935
+ }
936
+ //#endregion
937
+ export { readListing as _, PUBLIC_BASE_URL as a, name as c, DEFAULT_MAX_TOKENS as d, DEFAULT_STREAM_IDLE_TIMEOUT_MS as f, listingUrl as g, discoverChatModels as h, PROVIDER as i, resolveAdapterOptions as l, SiliconFlowAdapter as m, DEFAULT_API_KEY_ENV as n, apply as o, DISCOVERY_TTL_MS as p, DEFAULT_MODELS as r, inject as s, Config as t, DEFAULT_CONTEXT_WINDOW as u };