dsh-audiogen 0.4.6 → 0.4.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +489 -348
- package/lib/client.js.map +1 -1
- package/lib/index.js +94 -8
- package/package.json +1 -1
- package/src/client/SettingsCard.tsx +73 -15
- package/src/client/locales.ts +14 -0
- package/src/client/settings-scope.ts +2 -0
- package/src/client/studio-view.tsx +62 -8
- package/src/index.ts +56 -3
- package/src/prompt-enhance.ts +36 -6
- package/src/protocol.ts +16 -1
- package/src/routes.ts +20 -2
package/src/index.ts
CHANGED
|
@@ -16,7 +16,7 @@ import z from 'schemastery'
|
|
|
16
16
|
import type {} from '@deepseek-ai/dsh-host-webserver'
|
|
17
17
|
import type {} from '@deepseek-ai/dsh-system-prompt'
|
|
18
18
|
import type {} from '@deepseek-ai/dsh-tools'
|
|
19
|
-
import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type ModelMapping } from './protocol.ts'
|
|
19
|
+
import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type LlmModelOption, type ModelMapping } from './protocol.ts'
|
|
20
20
|
import { createGenerationBudget } from './audio-scheduler.ts'
|
|
21
21
|
import { enhancePromptText } from './prompt-enhance.ts'
|
|
22
22
|
import { AudioGenError } from './audio-engine.ts'
|
|
@@ -45,6 +45,11 @@ export interface Config {
|
|
|
45
45
|
autoSaveToLibrary?: boolean
|
|
46
46
|
/** 设置卡按文本编辑,保存值可能是数字或数字字符串。 */
|
|
47
47
|
maxConcurrentGenerations?: number | string
|
|
48
|
+
/**
|
|
49
|
+
* 提示词增强模型,格式 "provider|model"(如 "deepseek-official|deepseek-v4-flash-vision-exp");
|
|
50
|
+
* 空串表示跟随 Agent 默认模型(「设置 → 模型」)。
|
|
51
|
+
*/
|
|
52
|
+
enhanceModel?: string
|
|
48
53
|
}
|
|
49
54
|
|
|
50
55
|
const DEFAULT_MAX_CONCURRENT = 5
|
|
@@ -68,6 +73,7 @@ export const Config: z<Config> = z.object({
|
|
|
68
73
|
defaultModel: z.string().default(''),
|
|
69
74
|
autoSaveToLibrary: z.boolean().default(false),
|
|
70
75
|
maxConcurrentGenerations: z.union([z.number(), z.string()]).default(DEFAULT_MAX_CONCURRENT),
|
|
76
|
+
enhanceModel: z.string().default(''),
|
|
71
77
|
})
|
|
72
78
|
|
|
73
79
|
const DEFAULT_ENABLED = true
|
|
@@ -158,6 +164,8 @@ export interface EffectiveConfig {
|
|
|
158
164
|
defaultChannelId: string
|
|
159
165
|
defaultModel: string
|
|
160
166
|
autoSaveToLibrary: boolean
|
|
167
|
+
/** 提示词增强模型("provider|model";空串 = 跟随 Agent 默认模型)。 */
|
|
168
|
+
enhanceModel: string
|
|
161
169
|
maxConcurrentGenerations: number
|
|
162
170
|
}
|
|
163
171
|
|
|
@@ -186,6 +194,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
186
194
|
})),
|
|
187
195
|
defaultChannelId,
|
|
188
196
|
defaultModel: typeof value.defaultModel === 'string' ? value.defaultModel.trim() : '',
|
|
197
|
+
enhanceModel: typeof value.enhanceModel === 'string' ? value.enhanceModel.trim() : '',
|
|
189
198
|
autoSaveToLibrary: value.autoSaveToLibrary === true,
|
|
190
199
|
maxConcurrentGenerations: (() => {
|
|
191
200
|
const rawMax = value.maxConcurrentGenerations
|
|
@@ -198,11 +207,54 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
198
207
|
// 全局并发闸门:所有上游调用(面板路由 + Agent 工具)共享「最大并发生成数」。
|
|
199
208
|
const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations)
|
|
200
209
|
|
|
201
|
-
//
|
|
210
|
+
// 提示词增强:优先用设置的增强模型(enhanceModel),否则复用 Agent 默认模型
|
|
211
|
+
// (agent-default-model 设置);面板与 Agent 工具共用。
|
|
202
212
|
const enhance = async (prompt: string, mode: AudioMode): Promise<string> => {
|
|
203
213
|
const seam = ctx.get('settings') as unknown as SettingsSeam
|
|
204
214
|
if (seam?.describe === undefined) throw new AudioGenError('设置服务不可用,无法增强提示词', 'settings-unavailable')
|
|
205
|
-
return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode)
|
|
215
|
+
return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode, enhanceSelectionOf(resolve()))
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/** 解析设置的增强模型("provider|model");空/非法值返回 undefined = 跟随默认。 */
|
|
219
|
+
const enhanceSelectionOf = (config: EffectiveConfig): { provider: string; model: string } | undefined => {
|
|
220
|
+
const raw = config.enhanceModel ?? ''
|
|
221
|
+
const sep = raw.indexOf('|')
|
|
222
|
+
if (sep <= 0 || sep >= raw.length - 1) return undefined
|
|
223
|
+
const provider = raw.slice(0, sep).trim()
|
|
224
|
+
const model = raw.slice(sep + 1).trim()
|
|
225
|
+
return provider !== '' && model !== '' ? { provider, model } : undefined
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/** 读取「设置 → 模型」目录:各提供方 + 可广播的模型列表(增强模型下拉候选)。 */
|
|
229
|
+
const llmModelOptions = async (): Promise<LlmModelOption[]> => {
|
|
230
|
+
const llm = ctx.get('llm') as {
|
|
231
|
+
listProviders?: () => Array<{ id?: string; name?: string }>
|
|
232
|
+
listConfigurableProviders?: () => Array<{ provider?: string; displayName?: string }>
|
|
233
|
+
listModels?: (provider: string) => Promise<Array<{ id?: string; name?: string }>>
|
|
234
|
+
} | undefined
|
|
235
|
+
if (llm === undefined || llm.listProviders === undefined || llm.listModels === undefined) return []
|
|
236
|
+
const options: LlmModelOption[] = []
|
|
237
|
+
const directory = new Map<string, string>()
|
|
238
|
+
if (llm.listConfigurableProviders !== undefined) {
|
|
239
|
+
for (const entry of llm.listConfigurableProviders() ?? []) {
|
|
240
|
+
if (typeof entry.provider === 'string' && entry.provider !== '' && typeof entry.displayName === 'string') {
|
|
241
|
+
directory.set(entry.provider, entry.displayName)
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
for (const info of llm.listProviders() ?? []) {
|
|
246
|
+
const provider = typeof info.id === 'string' ? info.id : ''
|
|
247
|
+
if (provider === '') continue
|
|
248
|
+
let models: Array<{ id?: string; name?: string }> = []
|
|
249
|
+
try { models = (await llm.listModels(provider)) ?? [] } catch { models = [] }
|
|
250
|
+
for (const model of models) {
|
|
251
|
+
const id = typeof model.id === 'string' ? model.id.trim() : ''
|
|
252
|
+
if (id === '') continue
|
|
253
|
+
const name = typeof model.name === 'string' && model.name.trim() !== '' ? model.name.trim() : id
|
|
254
|
+
options.push({ provider, providerName: directory.get(provider) ?? provider, id, name })
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
return options
|
|
206
258
|
}
|
|
207
259
|
|
|
208
260
|
const channelsView = (): ChannelsView => {
|
|
@@ -219,6 +271,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
219
271
|
autoSave: () => resolve().autoSaveToLibrary,
|
|
220
272
|
budget,
|
|
221
273
|
enhance,
|
|
274
|
+
llmModelOptions,
|
|
222
275
|
})
|
|
223
276
|
const disposers = routes.map(route => ctx.webServer.register(route))
|
|
224
277
|
return () => { for (const dispose of disposers) dispose() }
|
package/src/prompt-enhance.ts
CHANGED
|
@@ -34,12 +34,29 @@ export interface PromptEnhanceDeps {
|
|
|
34
34
|
llm?: () => unknown
|
|
35
35
|
}
|
|
36
36
|
|
|
37
|
-
/**
|
|
38
|
-
export
|
|
37
|
+
/** 设置中显式选择的增强模型(provider + model)。 */
|
|
38
|
+
export interface EnhanceModelSelection {
|
|
39
|
+
provider: string
|
|
40
|
+
model: string
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。
|
|
44
|
+
* @param override - 用户设置的增强模型;缺省/为空时回退到 agent-default-model。 */
|
|
45
|
+
export async function enhancePromptText(
|
|
46
|
+
deps: PromptEnhanceDeps,
|
|
47
|
+
prompt: string,
|
|
48
|
+
mode: AudioMode,
|
|
49
|
+
override?: EnhanceModelSelection,
|
|
50
|
+
): Promise<string> {
|
|
39
51
|
const text = prompt.trim()
|
|
40
52
|
if (text === '') throw new AudioGenError('提示词为空,无法增强', 'enhance-empty-prompt')
|
|
41
|
-
const
|
|
42
|
-
|
|
53
|
+
const selected = override !== undefined && override.provider.trim() !== '' && override.model.trim() !== ''
|
|
54
|
+
? { provider: override.provider.trim(), model: override.model.trim() }
|
|
55
|
+
: undefined
|
|
56
|
+
const descriptor = selected === undefined
|
|
57
|
+
? (deps.settings.describe({ redactSecrets: true }) ?? []).find(candidate => String(candidate.ns) === 'agent-default-model')
|
|
58
|
+
: undefined
|
|
59
|
+
const value = selected ?? ((descriptor?.value ?? {}) as { provider?: unknown; model?: unknown })
|
|
43
60
|
const provider = typeof value.provider === 'string' && value.provider.trim() !== '' ? value.provider.trim() : ''
|
|
44
61
|
const model = typeof value.model === 'string' && value.model.trim() !== '' ? value.model.trim() : ''
|
|
45
62
|
if (provider === '' || model === '') {
|
|
@@ -53,22 +70,32 @@ export async function enhancePromptText(deps: PromptEnhanceDeps, prompt: string,
|
|
|
53
70
|
const timer = setTimeout(() => controller.abort(new DOMException('The operation timed out.', 'TimeoutError')), 30_000)
|
|
54
71
|
timer.unref?.()
|
|
55
72
|
let output = ''
|
|
73
|
+
let terminalFailure = ''
|
|
56
74
|
try {
|
|
57
75
|
for await (const chunk of runtime.stream({
|
|
58
76
|
provider,
|
|
59
77
|
model,
|
|
60
|
-
|
|
78
|
+
// dsh-llm 的 Message.content 是 ContentBlock[](如 [{ type: 'text', text }]),不是纯字符串;
|
|
79
|
+
// 传字符串会在适配器 contentHasImage() 里抛 "content.some is not a function",
|
|
80
|
+
// 主机把该异常转成终止的 finish/error 分片,导致下面永远收集不到文本。
|
|
81
|
+
messages: [{ role: 'user', content: [{ type: 'text', text }] }],
|
|
61
82
|
system: instructionsFor(mode),
|
|
62
83
|
temperature: 0.7,
|
|
63
84
|
maxTokens: 1200,
|
|
64
85
|
signal: controller.signal,
|
|
65
86
|
})) {
|
|
66
|
-
const record = chunk as { type?: string; text?: string; block?: { type?: string; text?: string } }
|
|
87
|
+
const record = chunk as { type?: string; text?: string; block?: { type?: string; text?: string }; reason?: { kind?: string; failure?: { message?: string; code?: string } } }
|
|
67
88
|
if (record.type === 'text-delta' && typeof record.text === 'string') {
|
|
68
89
|
output += record.text
|
|
69
90
|
} else if (record.type === 'block-end' && record.block !== undefined
|
|
70
91
|
&& record.block.type === 'text' && typeof record.block.text === 'string') {
|
|
71
92
|
output += record.block.text
|
|
93
|
+
} else if (record.type === 'finish' && record.reason !== undefined
|
|
94
|
+
&& record.reason.kind !== 'stop' && record.reason.kind !== undefined && terminalFailure === '') {
|
|
95
|
+
const failure = record.reason.failure
|
|
96
|
+
terminalFailure = typeof failure?.message === 'string' && failure.message.trim() !== ''
|
|
97
|
+
? `${failure.message}${typeof failure.code === 'string' ? `(${failure.code})` : ''}`
|
|
98
|
+
: `stream ${record.reason.kind}`
|
|
72
99
|
}
|
|
73
100
|
}
|
|
74
101
|
} finally {
|
|
@@ -76,6 +103,9 @@ export async function enhancePromptText(deps: PromptEnhanceDeps, prompt: string,
|
|
|
76
103
|
}
|
|
77
104
|
const result = stripFences(output.trim())
|
|
78
105
|
if (result === '') {
|
|
106
|
+
if (terminalFailure !== '') {
|
|
107
|
+
throw new AudioGenError(`增强失败:LLM 调用出错(${terminalFailure})。请检查「设置 → 模型」的默认模型是否可用`, 'enhance-llm-error')
|
|
108
|
+
}
|
|
79
109
|
throw new AudioGenError('模型未返回增强内容:请检查「设置 → 模型」的默认模型是否可用(或稍后重试)', 'enhance-empty-result')
|
|
80
110
|
}
|
|
81
111
|
return result
|
package/src/protocol.ts
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
|
|
9
9
|
|
|
10
10
|
/** Published package version shared by the host updater and the client UI. */
|
|
11
|
-
export const PLUGIN_VERSION = '0.4.
|
|
11
|
+
export const PLUGIN_VERSION = '0.4.8'
|
|
12
12
|
|
|
13
13
|
/** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
|
|
14
14
|
export const SETTINGS_API = {
|
|
@@ -30,6 +30,9 @@ export const ENHANCE_API = '/api/dsh-audiogen/prompt/enhance' as const
|
|
|
30
30
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
31
31
|
export const PRESETS_API = '/api/dsh-audiogen/presets' as const
|
|
32
32
|
|
|
33
|
+
/** LLM 模型目录:提示词增强模型的候选(来自「设置 → 模型」各提供方)。 */
|
|
34
|
+
export const LLM_MODELS_API = '/api/dsh-audiogen/llm/models' as const
|
|
35
|
+
|
|
33
36
|
/** Host-mediated model/voice discovery endpoint. */
|
|
34
37
|
export const MODEL_API = {
|
|
35
38
|
discover: '/api/dsh-audiogen/models/discover',
|
|
@@ -91,6 +94,18 @@ export interface ModelMapping {
|
|
|
91
94
|
category?: AudioModelCategory
|
|
92
95
|
}
|
|
93
96
|
|
|
97
|
+
/** 一个「设置 → 模型」中的 LLM 候选模型(提示词增强模型下拉用)。 */
|
|
98
|
+
export interface LlmModelOption {
|
|
99
|
+
/** 提供方路由(如 deepseek-official / google)。 */
|
|
100
|
+
provider: string
|
|
101
|
+
/** 提供方展示名(如 DeepSeek / google)。 */
|
|
102
|
+
providerName: string
|
|
103
|
+
/** 模型 id(上游型号,如 deepseek-v4-flash-vision-exp)。 */
|
|
104
|
+
id: string
|
|
105
|
+
/** 模型展示名。 */
|
|
106
|
+
name: string
|
|
107
|
+
}
|
|
108
|
+
|
|
94
109
|
/** A model/voice discovered from a vendor endpoint (not yet persisted). */
|
|
95
110
|
export interface DiscoveredAudioModel extends ModelMapping {
|
|
96
111
|
/** Optional human-readable description from the vendor. */
|
package/src/routes.ts
CHANGED
|
@@ -16,9 +16,9 @@ import { discoverAudioModels } from './audio-models.ts'
|
|
|
16
16
|
import { AUDIO_PRESETS } from './audio-presets.ts'
|
|
17
17
|
import { appendHistory, clearHistory, listHistory, readAudioFile, removeHistory, saveAudioFile, listLibrary, saveToLibrary, updateLibraryEntry, removeLibraryEntries, readLibraryFile } from './audio-store.ts'
|
|
18
18
|
import {
|
|
19
|
-
AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
|
|
19
|
+
AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, LLM_MODELS_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
|
|
20
20
|
LIBRARY_TYPES,
|
|
21
|
-
type GenerateAudioRequest, type GeneratedAudio, type HistoryEntryInput, type LibraryAudioInput, type LibraryProvenance, type LibraryType,
|
|
21
|
+
type GenerateAudioRequest, type GeneratedAudio, type HistoryEntryInput, type LibraryAudioInput, type LibraryProvenance, type LibraryType, type LlmModelOption,
|
|
22
22
|
} from './protocol.ts'
|
|
23
23
|
|
|
24
24
|
const MAX_JSON_BODY_BYTES = 16 * 1024 * 1024
|
|
@@ -49,6 +49,8 @@ export interface AudiogenRoutesDeps {
|
|
|
49
49
|
budget: GenerationBudget
|
|
50
50
|
/** 提示词增强:调用 Agent 默认模型,返回增强后的文本。 */
|
|
51
51
|
enhance: (prompt: string, mode: GenerateAudioRequest['mode']) => Promise<string>
|
|
52
|
+
/** 「设置 → 模型」提供方列表 + 各自可广播模型(增强模型下拉候选)。 */
|
|
53
|
+
llmModelOptions: () => Promise<LlmModelOption[]>
|
|
52
54
|
}
|
|
53
55
|
|
|
54
56
|
function isLoopbackRequest(request: IncomingMessage): boolean {
|
|
@@ -142,6 +144,8 @@ function parseGenerateRequest(body: Record<string, unknown>): GenerateAudioReque
|
|
|
142
144
|
...(num(body.cfgScale) !== undefined ? { cfgScale: num(body.cfgScale)! } : {}),
|
|
143
145
|
...(typeof body.format === 'string' && body.format.trim() !== '' ? { format: body.format.trim() } : {}),
|
|
144
146
|
...(typeof body.channelId === 'string' && body.channelId !== '' ? { channelId: body.channelId } : {}),
|
|
147
|
+
// 对比任务的任务 id:随 params 一起入历史,供客户端按 taskId 聚合出对比卡片。
|
|
148
|
+
...(typeof body.taskId === 'string' && body.taskId.trim() !== '' ? { taskId: body.taskId.trim() } : {}),
|
|
145
149
|
// ---- MiniMax TTS 专属字段(其他厂商忽略) ----
|
|
146
150
|
...(str(body.emotion) !== undefined ? { emotion: str(body.emotion)! } : {}),
|
|
147
151
|
...(num(body.vol) !== undefined ? { vol: num(body.vol)! } : {}),
|
|
@@ -327,6 +331,20 @@ export function makeRoutes(deps: AudiogenRoutesDeps): WebRoute[] {
|
|
|
327
331
|
writeJson(res, 200, { ok: true, presets: AUDIO_PRESETS })
|
|
328
332
|
},
|
|
329
333
|
},
|
|
334
|
+
// --------------------------------------------- LLM models (enhance)
|
|
335
|
+
{
|
|
336
|
+
kind: 'exact',
|
|
337
|
+
path: LLM_MODELS_API,
|
|
338
|
+
handler: async (req, res) => {
|
|
339
|
+
if (!guard(req, res, 'POST')) return
|
|
340
|
+
try {
|
|
341
|
+
const providers = await deps.llmModelOptions()
|
|
342
|
+
writeJson(res, 200, { ok: true, providers })
|
|
343
|
+
} catch (error) {
|
|
344
|
+
writeJson(res, 200, { ok: false, code: 'llm-models-failed', message: messageOf(error) })
|
|
345
|
+
}
|
|
346
|
+
},
|
|
347
|
+
},
|
|
330
348
|
// ---------------------------------------------- model/voice discovery
|
|
331
349
|
{
|
|
332
350
|
kind: 'exact',
|