dsh-audiogen 0.4.6 → 0.4.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +474 -352
- package/lib/client.js.map +1 -1
- package/lib/index.js +93 -8
- package/package.json +1 -1
- package/src/client/SettingsCard.tsx +73 -15
- package/src/client/locales.ts +14 -0
- package/src/client/settings-scope.ts +2 -0
- package/src/client/studio-view.tsx +38 -6
- package/src/index.ts +56 -3
- package/src/prompt-enhance.ts +36 -6
- package/src/protocol.ts +16 -1
- package/src/routes.ts +18 -2
package/lib/index.js
CHANGED
|
@@ -28,6 +28,8 @@ const TASK_API = { cancel: "/api/dsh-audiogen/task/cancel" };
|
|
|
28
28
|
const ENHANCE_API = "/api/dsh-audiogen/prompt/enhance";
|
|
29
29
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
30
30
|
const PRESETS_API = "/api/dsh-audiogen/presets";
|
|
31
|
+
/** LLM 模型目录:提示词增强模型的候选(来自「设置 → 模型」各提供方)。 */
|
|
32
|
+
const LLM_MODELS_API = "/api/dsh-audiogen/llm/models";
|
|
31
33
|
/** Host-mediated model/voice discovery endpoint. */
|
|
32
34
|
const MODEL_API = { discover: "/api/dsh-audiogen/models/discover" };
|
|
33
35
|
/** Loopback-only audio file reader for panel/tool-result previews. */
|
|
@@ -882,11 +884,17 @@ function instructionsFor(mode) {
|
|
|
882
884
|
voice_design: "这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。"
|
|
883
885
|
}[mode];
|
|
884
886
|
}
|
|
885
|
-
/** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。
|
|
886
|
-
|
|
887
|
+
/** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。
|
|
888
|
+
* @param override - 用户设置的增强模型;缺省/为空时回退到 agent-default-model。 */
|
|
889
|
+
async function enhancePromptText(deps, prompt, mode, override) {
|
|
887
890
|
const text = prompt.trim();
|
|
888
891
|
if (text === "") throw new AudioGenError("提示词为空,无法增强", "enhance-empty-prompt");
|
|
889
|
-
const
|
|
892
|
+
const selected = override !== void 0 && override.provider.trim() !== "" && override.model.trim() !== "" ? {
|
|
893
|
+
provider: override.provider.trim(),
|
|
894
|
+
model: override.model.trim()
|
|
895
|
+
} : void 0;
|
|
896
|
+
const descriptor = selected === void 0 ? (deps.settings.describe({ redactSecrets: true }) ?? []).find((candidate) => String(candidate.ns) === "agent-default-model") : void 0;
|
|
897
|
+
const value = selected ?? descriptor?.value ?? {};
|
|
890
898
|
const provider = typeof value.provider === "string" && value.provider.trim() !== "" ? value.provider.trim() : "";
|
|
891
899
|
const model = typeof value.model === "string" && value.model.trim() !== "" ? value.model.trim() : "";
|
|
892
900
|
if (provider === "" || model === "") throw new AudioGenError("未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型", "no-default-model");
|
|
@@ -896,13 +904,17 @@ async function enhancePromptText(deps, prompt, mode) {
|
|
|
896
904
|
const timer = setTimeout(() => controller.abort(new DOMException("The operation timed out.", "TimeoutError")), 3e4);
|
|
897
905
|
timer.unref?.();
|
|
898
906
|
let output = "";
|
|
907
|
+
let terminalFailure = "";
|
|
899
908
|
try {
|
|
900
909
|
for await (const chunk of runtime.stream({
|
|
901
910
|
provider,
|
|
902
911
|
model,
|
|
903
912
|
messages: [{
|
|
904
913
|
role: "user",
|
|
905
|
-
content:
|
|
914
|
+
content: [{
|
|
915
|
+
type: "text",
|
|
916
|
+
text
|
|
917
|
+
}]
|
|
906
918
|
}],
|
|
907
919
|
system: instructionsFor(mode),
|
|
908
920
|
temperature: .7,
|
|
@@ -912,12 +924,19 @@ async function enhancePromptText(deps, prompt, mode) {
|
|
|
912
924
|
const record = chunk;
|
|
913
925
|
if (record.type === "text-delta" && typeof record.text === "string") output += record.text;
|
|
914
926
|
else if (record.type === "block-end" && record.block !== void 0 && record.block.type === "text" && typeof record.block.text === "string") output += record.block.text;
|
|
927
|
+
else if (record.type === "finish" && record.reason !== void 0 && record.reason.kind !== "stop" && record.reason.kind !== void 0 && terminalFailure === "") {
|
|
928
|
+
const failure = record.reason.failure;
|
|
929
|
+
terminalFailure = typeof failure?.message === "string" && failure.message.trim() !== "" ? `${failure.message}${typeof failure.code === "string" ? `(${failure.code})` : ""}` : `stream ${record.reason.kind}`;
|
|
930
|
+
}
|
|
915
931
|
}
|
|
916
932
|
} finally {
|
|
917
933
|
clearTimeout(timer);
|
|
918
934
|
}
|
|
919
935
|
const result = stripFences(output.trim());
|
|
920
|
-
if (result === "")
|
|
936
|
+
if (result === "") {
|
|
937
|
+
if (terminalFailure !== "") throw new AudioGenError(`增强失败:LLM 调用出错(${terminalFailure})。请检查「设置 → 模型」的默认模型是否可用`, "enhance-llm-error");
|
|
938
|
+
throw new AudioGenError("模型未返回增强内容:请检查「设置 → 模型」的默认模型是否可用(或稍后重试)", "enhance-empty-result");
|
|
939
|
+
}
|
|
921
940
|
return result;
|
|
922
941
|
}
|
|
923
942
|
/** 去掉模型可能包裹的 ``` 代码围栏。 */
|
|
@@ -1871,6 +1890,25 @@ function makeRoutes(deps) {
|
|
|
1871
1890
|
});
|
|
1872
1891
|
}
|
|
1873
1892
|
},
|
|
1893
|
+
{
|
|
1894
|
+
kind: "exact",
|
|
1895
|
+
path: LLM_MODELS_API,
|
|
1896
|
+
handler: async (req, res) => {
|
|
1897
|
+
if (!guard(req, res, "POST")) return;
|
|
1898
|
+
try {
|
|
1899
|
+
writeJson(res, 200, {
|
|
1900
|
+
ok: true,
|
|
1901
|
+
providers: await deps.llmModelOptions()
|
|
1902
|
+
});
|
|
1903
|
+
} catch (error) {
|
|
1904
|
+
writeJson(res, 200, {
|
|
1905
|
+
ok: false,
|
|
1906
|
+
code: "llm-models-failed",
|
|
1907
|
+
message: messageOf(error)
|
|
1908
|
+
});
|
|
1909
|
+
}
|
|
1910
|
+
}
|
|
1911
|
+
},
|
|
1874
1912
|
{
|
|
1875
1913
|
kind: "exact",
|
|
1876
1914
|
path: MODEL_API.discover,
|
|
@@ -3134,7 +3172,8 @@ const Config = z.object({
|
|
|
3134
3172
|
defaultChannelId: z.string().default(""),
|
|
3135
3173
|
defaultModel: z.string().default(""),
|
|
3136
3174
|
autoSaveToLibrary: z.boolean().default(false),
|
|
3137
|
-
maxConcurrentGenerations: z.union([z.number(), z.string()]).default(DEFAULT_MAX_CONCURRENT)
|
|
3175
|
+
maxConcurrentGenerations: z.union([z.number(), z.string()]).default(DEFAULT_MAX_CONCURRENT),
|
|
3176
|
+
enhanceModel: z.string().default("")
|
|
3138
3177
|
});
|
|
3139
3178
|
const DEFAULT_ENABLED = true;
|
|
3140
3179
|
const DEFAULT_ANNOUNCE = true;
|
|
@@ -3227,6 +3266,7 @@ function apply(ctx, config) {
|
|
|
3227
3266
|
})),
|
|
3228
3267
|
defaultChannelId,
|
|
3229
3268
|
defaultModel: typeof value.defaultModel === "string" ? value.defaultModel.trim() : "",
|
|
3269
|
+
enhanceModel: typeof value.enhanceModel === "string" ? value.enhanceModel.trim() : "",
|
|
3230
3270
|
autoSaveToLibrary: value.autoSaveToLibrary === true,
|
|
3231
3271
|
maxConcurrentGenerations: (() => {
|
|
3232
3272
|
const rawMax = value.maxConcurrentGenerations;
|
|
@@ -3242,7 +3282,51 @@ function apply(ctx, config) {
|
|
|
3242
3282
|
return enhancePromptText({
|
|
3243
3283
|
settings: seam,
|
|
3244
3284
|
llm: () => ctx.get("llm")
|
|
3245
|
-
}, prompt, mode);
|
|
3285
|
+
}, prompt, mode, enhanceSelectionOf(resolve()));
|
|
3286
|
+
};
|
|
3287
|
+
/** 解析设置的增强模型("provider|model");空/非法值返回 undefined = 跟随默认。 */
|
|
3288
|
+
const enhanceSelectionOf = (config) => {
|
|
3289
|
+
const raw = config.enhanceModel ?? "";
|
|
3290
|
+
const sep = raw.indexOf("|");
|
|
3291
|
+
if (sep <= 0 || sep >= raw.length - 1) return void 0;
|
|
3292
|
+
const provider = raw.slice(0, sep).trim();
|
|
3293
|
+
const model = raw.slice(sep + 1).trim();
|
|
3294
|
+
return provider !== "" && model !== "" ? {
|
|
3295
|
+
provider,
|
|
3296
|
+
model
|
|
3297
|
+
} : void 0;
|
|
3298
|
+
};
|
|
3299
|
+
/** 读取「设置 → 模型」目录:各提供方 + 可广播的模型列表(增强模型下拉候选)。 */
|
|
3300
|
+
const llmModelOptions = async () => {
|
|
3301
|
+
const llm = ctx.get("llm");
|
|
3302
|
+
if (llm === void 0 || llm.listProviders === void 0 || llm.listModels === void 0) return [];
|
|
3303
|
+
const options = [];
|
|
3304
|
+
const directory = /* @__PURE__ */ new Map();
|
|
3305
|
+
if (llm.listConfigurableProviders !== void 0) {
|
|
3306
|
+
for (const entry of llm.listConfigurableProviders() ?? []) if (typeof entry.provider === "string" && entry.provider !== "" && typeof entry.displayName === "string") directory.set(entry.provider, entry.displayName);
|
|
3307
|
+
}
|
|
3308
|
+
for (const info of llm.listProviders() ?? []) {
|
|
3309
|
+
const provider = typeof info.id === "string" ? info.id : "";
|
|
3310
|
+
if (provider === "") continue;
|
|
3311
|
+
let models = [];
|
|
3312
|
+
try {
|
|
3313
|
+
models = await llm.listModels(provider) ?? [];
|
|
3314
|
+
} catch {
|
|
3315
|
+
models = [];
|
|
3316
|
+
}
|
|
3317
|
+
for (const model of models) {
|
|
3318
|
+
const id = typeof model.id === "string" ? model.id.trim() : "";
|
|
3319
|
+
if (id === "") continue;
|
|
3320
|
+
const name = typeof model.name === "string" && model.name.trim() !== "" ? model.name.trim() : id;
|
|
3321
|
+
options.push({
|
|
3322
|
+
provider,
|
|
3323
|
+
providerName: directory.get(provider) ?? provider,
|
|
3324
|
+
id,
|
|
3325
|
+
name
|
|
3326
|
+
});
|
|
3327
|
+
}
|
|
3328
|
+
}
|
|
3329
|
+
return options;
|
|
3246
3330
|
};
|
|
3247
3331
|
const channelsView = () => {
|
|
3248
3332
|
const value = resolve();
|
|
@@ -3259,7 +3343,8 @@ function apply(ctx, config) {
|
|
|
3259
3343
|
resolveChannels: channelsView,
|
|
3260
3344
|
autoSave: () => resolve().autoSaveToLibrary,
|
|
3261
3345
|
budget,
|
|
3262
|
-
enhance
|
|
3346
|
+
enhance,
|
|
3347
|
+
llmModelOptions
|
|
3263
3348
|
}).map((route) => ctx.webServer.register(route));
|
|
3264
3349
|
return () => {
|
|
3265
3350
|
for (const dispose of disposers) dispose();
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-audiogen",
|
|
3
3
|
"description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.7",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"exports": {
|
|
@@ -18,7 +18,7 @@ import { createSnapshotStore, type SnapshotStore } from '@deepseek-ai/dsh-client
|
|
|
18
18
|
import { CardForm, booleanField, textField, type CardActions, type CardShell, type FieldState as CardFieldState } from './settings-form.ts'
|
|
19
19
|
import { ChannelsForm, type ChannelDraft, type ChannelsFormActions, type ChannelsFormState } from './channels-form.ts'
|
|
20
20
|
import type { AudiogenScope } from './settings-scope.ts'
|
|
21
|
-
import { MODEL_API, PRESETS_API, type AudioModelCategory, type DiscoveredAudioModel, type ModelMapping, type PresetProviderView } from '../protocol.ts'
|
|
21
|
+
import { MODEL_API, PRESETS_API, LLM_MODELS_API, type AudioModelCategory, type DiscoveredAudioModel, type LlmModelOption, type ModelMapping, type PresetProviderView } from '../protocol.ts'
|
|
22
22
|
import type { AudioGenKey } from './locales.ts'
|
|
23
23
|
import css from './settings-card.module.css'
|
|
24
24
|
|
|
@@ -29,6 +29,7 @@ export interface AudioGenSettings {
|
|
|
29
29
|
defaultModel?: string
|
|
30
30
|
autoSaveToLibrary?: boolean
|
|
31
31
|
maxConcurrentGenerations?: number
|
|
32
|
+
enhanceModel?: string
|
|
32
33
|
}
|
|
33
34
|
|
|
34
35
|
export interface AudioGenSettingsCardState extends CardShell {
|
|
@@ -39,6 +40,7 @@ export interface AudioGenSettingsCardState extends CardShell {
|
|
|
39
40
|
defaultModel: CardFieldState
|
|
40
41
|
autoSaveToLibrary: CardFieldState
|
|
41
42
|
maxConcurrentGenerations: CardFieldState
|
|
43
|
+
enhanceModel: CardFieldState
|
|
42
44
|
}
|
|
43
45
|
|
|
44
46
|
export interface AudioGenSettingsCardFace extends CardActions {
|
|
@@ -60,6 +62,7 @@ export class AudioGenSettingsCardController {
|
|
|
60
62
|
textField('defaultModel'),
|
|
61
63
|
booleanField('autoSaveToLibrary'),
|
|
62
64
|
textField('maxConcurrentGenerations'),
|
|
65
|
+
textField('enhanceModel'),
|
|
63
66
|
])
|
|
64
67
|
this.channelsForm = new ChannelsForm(scope)
|
|
65
68
|
}
|
|
@@ -76,6 +79,7 @@ export class AudioGenSettingsCardController {
|
|
|
76
79
|
defaultModel: this.form.field('defaultModel'),
|
|
77
80
|
autoSaveToLibrary: this.form.field('autoSaveToLibrary'),
|
|
78
81
|
maxConcurrentGenerations: this.form.field('maxConcurrentGenerations'),
|
|
82
|
+
enhanceModel: this.form.field('enhanceModel'),
|
|
79
83
|
}
|
|
80
84
|
}
|
|
81
85
|
|
|
@@ -387,6 +391,9 @@ function ChannelEditor(props: ChannelEditorProps): React.JSX.Element {
|
|
|
387
391
|
onAdopt={() => adoptCandidates()}
|
|
388
392
|
onCloseCandidates={closeCandidates}
|
|
389
393
|
/>
|
|
394
|
+
{/stability/i.test(`${props.channel?.preset ?? invited?.id ?? presetId}|${url}`) ? (
|
|
395
|
+
<p className={css.sectionHint}>{t('channel.stabilityHint')}</p>
|
|
396
|
+
) : null}
|
|
390
397
|
<label className={css.field}>
|
|
391
398
|
<span className={css.label}>
|
|
392
399
|
<input type="checkbox" checked={isDefault} disabled={!writable} onChange={event => setIsDefault(event.target.checked)} /> {t('channel.default')}
|
|
@@ -469,20 +476,32 @@ function ModelCatalog(props: ModelCatalogProps): React.JSX.Element {
|
|
|
469
476
|
disabled={!writable}
|
|
470
477
|
onChange={event => props.onPatchModel(index, { id: event.target.value })}
|
|
471
478
|
/>
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
479
|
+
{/^stable-audio-/i.test(model.id) ? (
|
|
480
|
+
<select
|
|
481
|
+
className={`${css.select} ${css.modelCategorySelect}`}
|
|
482
|
+
value="__stable-dual__"
|
|
483
|
+
aria-label={`${t('channel.modelCategory')} ${index + 1}`}
|
|
484
|
+
disabled
|
|
485
|
+
title={t('channel.category.stableDualHint')}
|
|
486
|
+
>
|
|
487
|
+
<option value="__stable-dual__">{t('channel.category.stableDual')}</option>
|
|
488
|
+
</select>
|
|
489
|
+
) : (
|
|
490
|
+
<select
|
|
491
|
+
className={`${css.select} ${css.modelCategorySelect}`}
|
|
492
|
+
value={model.category ?? ''}
|
|
493
|
+
aria-label={`${t('channel.modelCategory')} ${index + 1}`}
|
|
494
|
+
disabled={!writable}
|
|
495
|
+
onChange={event => {
|
|
496
|
+
const value = event.target.value as AudioModelCategory | ''
|
|
497
|
+
props.onPatchModel(index, value === '' ? { category: undefined } : { category: value })
|
|
498
|
+
}}
|
|
499
|
+
>
|
|
500
|
+
{MODEL_CATEGORIES.map(category => (
|
|
501
|
+
<option key={category ?? 'auto'} value={category ?? ''}>{category === undefined ? t('channel.category.auto') : t(`channel.category.${category}`)}</option>
|
|
502
|
+
))}
|
|
503
|
+
</select>
|
|
504
|
+
)}
|
|
486
505
|
<button
|
|
487
506
|
type="button"
|
|
488
507
|
className={css.modelRowRemove}
|
|
@@ -554,6 +573,8 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
|
|
|
554
573
|
const [presetLoading, setPresetLoading] = useState(false)
|
|
555
574
|
const [presetError, setPresetError] = useState<string | null>(null)
|
|
556
575
|
const [confirmDeleteId, setConfirmDeleteId] = useState<string | null>(null)
|
|
576
|
+
const [llmModels, setLlmModels] = useState<LlmModelOption[] | null>(null)
|
|
577
|
+
const [llmModelsError, setLlmModelsError] = useState<string | null>(null)
|
|
557
578
|
|
|
558
579
|
const channels = state.channels.channels
|
|
559
580
|
const editingChannel = editor !== null && editor.kind === 'edit'
|
|
@@ -575,6 +596,18 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
|
|
|
575
596
|
}
|
|
576
597
|
}, [editor, presets.length, presetError, presetLoading])
|
|
577
598
|
|
|
599
|
+
// 展开卡片时加载「设置 → 模型」的 LLM 提供方/模型列表(提示词增强模型下拉)。
|
|
600
|
+
useEffect(() => {
|
|
601
|
+
if (!open || llmModels !== null || llmModelsError !== null) return
|
|
602
|
+
void fetch(LLM_MODELS_API, { method: 'POST' })
|
|
603
|
+
.then(async response => {
|
|
604
|
+
const body = await response.json() as { ok?: boolean; providers?: LlmModelOption[]; message?: string }
|
|
605
|
+
if (!response.ok || body.ok !== true || body.providers === undefined) throw new Error(body.message ?? `HTTP ${response.status}`)
|
|
606
|
+
setLlmModels(body.providers)
|
|
607
|
+
})
|
|
608
|
+
.catch(error => setLlmModelsError(error instanceof Error ? error.message : String(error)))
|
|
609
|
+
}, [open, llmModels, llmModelsError])
|
|
610
|
+
|
|
578
611
|
if (!state.available) return null
|
|
579
612
|
|
|
580
613
|
const blocked = !state.dirty || state.invalid || state.saving || state.channels.saving
|
|
@@ -707,6 +740,31 @@ export function AudioGenSettingsCard(props: AudioGenSettingsCardProps) {
|
|
|
707
740
|
) : null}
|
|
708
741
|
</section>
|
|
709
742
|
|
|
743
|
+
<div className={css.field}>
|
|
744
|
+
<label className={css.label}>
|
|
745
|
+
<span>{t('settings.enhanceModel')}</span>
|
|
746
|
+
<select
|
|
747
|
+
className={css.select}
|
|
748
|
+
value={state.enhanceModel.text}
|
|
749
|
+
disabled={!state.writable}
|
|
750
|
+
onChange={event => props.edit('enhanceModel', event.target.value)}
|
|
751
|
+
>
|
|
752
|
+
<option value="">{t('settings.enhanceModelDefault')}</option>
|
|
753
|
+
{llmModels !== null ? llmModels.map(option => (
|
|
754
|
+
<option key={`${option.provider}|${option.id}`} value={`${option.provider}|${option.id}`}>
|
|
755
|
+
{`${option.providerName} — ${option.name}${option.name !== option.id ? `(${option.id})` : ''}`}
|
|
756
|
+
</option>
|
|
757
|
+
)) : null}
|
|
758
|
+
{llmModels !== null && state.enhanceModel.text !== ''
|
|
759
|
+
&& !llmModels.some(option => `${option.provider}|${option.id}` === state.enhanceModel.text) ? (
|
|
760
|
+
<option value={state.enhanceModel.text}>{state.enhanceModel.text}</option>
|
|
761
|
+
) : null}
|
|
762
|
+
</select>
|
|
763
|
+
<span className={css.sectionHint}>{t('settings.enhanceModelHint')}</span>
|
|
764
|
+
{llmModelsError !== null ? <span className={css.failed}>{t('settings.enhanceModelFailed')}:{llmModelsError}</span> : null}
|
|
765
|
+
</label>
|
|
766
|
+
</div>
|
|
767
|
+
|
|
710
768
|
<div className={css.field}>
|
|
711
769
|
<label className={css.label}>
|
|
712
770
|
<input type="checkbox" checked={state.enabled.text === 'true' || state.enabled.text === ''} disabled={!state.writable} onChange={event => props.edit('enabled', String(event.target.checked))} /> {t('settings.enabled')}
|
package/src/client/locales.ts
CHANGED
|
@@ -37,6 +37,10 @@ export const zh = {
|
|
|
37
37
|
'settings.allowAgentAudio': '允许 Agent 调用音频生成',
|
|
38
38
|
'settings.autoSaveLibrary': '生成后自动保存到资源库',
|
|
39
39
|
'settings.maxConcurrent': '最大并发生成数(同时打到上游的请求数,默认 5)',
|
|
40
|
+
'settings.enhanceModel': '提示词增强模型(LLM)',
|
|
41
|
+
'settings.enhanceModelDefault': '跟随 Agent 默认模型(设置 → 模型)',
|
|
42
|
+
'settings.enhanceModelHint': '用于「✨ 增强提示词」与 generate_audio 的 enhance_prompt 参数;留空则使用「设置 → 模型」中的默认模型。',
|
|
43
|
+
'settings.enhanceModelFailed': '模型列表加载失败',
|
|
40
44
|
'settings.save': '保存',
|
|
41
45
|
'settings.saving': '保存中…',
|
|
42
46
|
'settings.discard': '放弃修改',
|
|
@@ -89,6 +93,8 @@ export const zh = {
|
|
|
89
93
|
'channel.category.sfx': '音效',
|
|
90
94
|
'channel.category.voice_design': '音色设计',
|
|
91
95
|
'channel.category.voice_clone': '音色克隆',
|
|
96
|
+
'channel.category.stableDual': '音乐 + 音效(自动)',
|
|
97
|
+
'channel.category.stableDualHint': 'Stable Audio 模型按模型名自动识别为音乐+音效,无需设置分类。',
|
|
92
98
|
'channel.addModel': '添加模型',
|
|
93
99
|
'channel.removeModel': '移除',
|
|
94
100
|
'channel.fetchModels': '获取可用模型',
|
|
@@ -97,6 +103,7 @@ export const zh = {
|
|
|
97
103
|
'channel.fetchEmpty': '未发现音频相关模型。',
|
|
98
104
|
'channel.discoverSource': '来源:{source}',
|
|
99
105
|
'channel.candidates': '发现 {n} 个可用模型/音色',
|
|
106
|
+
'channel.stabilityHint': 'Stable Audio 模型(stable-audio-*)同时适用于「音乐生成」与「音效生成」,面板按模型名自动识别;分类仅影响默认展示。官方不支持 TTS。',
|
|
100
107
|
'channel.selectAll': '全选',
|
|
101
108
|
'channel.clearSelection': '清空',
|
|
102
109
|
'channel.adoptSelected': '添加所选({n})',
|
|
@@ -142,6 +149,10 @@ export const en: Record<AudioGenKey, string> = {
|
|
|
142
149
|
'settings.allowAgentAudio': 'Allow agents to generate audio',
|
|
143
150
|
'settings.autoSaveLibrary': 'Auto-save generated audio to the library',
|
|
144
151
|
'settings.maxConcurrent': 'Max concurrent generations (in-flight upstream calls, default 5)',
|
|
152
|
+
'settings.enhanceModel': 'Prompt enhance model (LLM)',
|
|
153
|
+
'settings.enhanceModelDefault': 'Follow the agent default model (Settings → Models)',
|
|
154
|
+
'settings.enhanceModelHint': 'Used by "✨ Enhance prompt" and the generate_audio enhance_prompt parameter; empty follows the default model in Settings → Models.',
|
|
155
|
+
'settings.enhanceModelFailed': 'Failed to load the model list',
|
|
145
156
|
'settings.save': 'Save',
|
|
146
157
|
'settings.saving': 'Saving…',
|
|
147
158
|
'settings.discard': 'Discard',
|
|
@@ -194,6 +205,8 @@ export const en: Record<AudioGenKey, string> = {
|
|
|
194
205
|
'channel.category.sfx': 'SFX',
|
|
195
206
|
'channel.category.voice_design': 'Voice design',
|
|
196
207
|
'channel.category.voice_clone': 'Voice clone',
|
|
208
|
+
'channel.category.stableDual': 'Music + SFX (auto)',
|
|
209
|
+
'channel.category.stableDualHint': 'Stable Audio models are auto-detected as Music + SFX by model name; no category needed.',
|
|
197
210
|
'channel.addModel': 'Add model',
|
|
198
211
|
'channel.removeModel': 'Remove',
|
|
199
212
|
'channel.fetchModels': 'Fetch available models',
|
|
@@ -202,6 +215,7 @@ export const en: Record<AudioGenKey, string> = {
|
|
|
202
215
|
'channel.fetchEmpty': 'No audio-related models found.',
|
|
203
216
|
'channel.discoverSource': 'Source: {source}',
|
|
204
217
|
'channel.candidates': '{n} available model(s)/voice(s) found',
|
|
218
|
+
'channel.stabilityHint': 'Stable Audio models (stable-audio-*) apply to both Music and SFX; the panel detects them by model name. The category only affects the default display. Official API does not support TTS.',
|
|
205
219
|
'channel.selectAll': 'Select all',
|
|
206
220
|
'channel.clearSelection': 'Clear',
|
|
207
221
|
'channel.adoptSelected': 'Add selected ({n})',
|
|
@@ -31,6 +31,8 @@ export interface AudiogenConfig {
|
|
|
31
31
|
defaultModel?: string
|
|
32
32
|
/** 生成完成后自动保存到资源库(面板与 Agent 生成均生效)。 */
|
|
33
33
|
autoSaveToLibrary?: boolean
|
|
34
|
+
/** 提示词增强模型("provider|model";空串 = 跟随 Agent 默认模型)。 */
|
|
35
|
+
enhanceModel?: string
|
|
34
36
|
}
|
|
35
37
|
|
|
36
38
|
/** One settings path-op as the bridge consumes it. */
|
|
@@ -16,11 +16,26 @@ import {
|
|
|
16
16
|
HISTORY_API,
|
|
17
17
|
type AudioMode, type GeneratedAudio, type GenerateAudioRequest, type HistoryEntry, type LibraryEntry,
|
|
18
18
|
} from '../protocol.ts'
|
|
19
|
+
import type { ModelOption } from './settings-scope.ts'
|
|
19
20
|
import { AudioPlayer } from './audio-player.tsx'
|
|
20
21
|
import { LibrarySaveDialog, type SaveDialogContext } from './library-save-dialog.tsx'
|
|
21
22
|
import { CheckIcon, DownloadIcon, StarIcon } from './icons.tsx'
|
|
22
23
|
import css from './audio-panel.module.css'
|
|
23
24
|
|
|
25
|
+
/**
|
|
26
|
+
* 模型是否适用于当前模式。
|
|
27
|
+
* - stable-audio-*(Stable Audio 系列):官方 text-to-audio 协议对音乐与音效是同一接口,
|
|
28
|
+
* 因此同时适用于「音乐生成」与「音效生成」;官方不支持 TTS(语音合成),故不出现在 TTS。
|
|
29
|
+
* - 其余模型:按设置中的分类(category)匹配,auto(未分类)适用于全部模式。
|
|
30
|
+
*/
|
|
31
|
+
function modelSuitableForMode(entry: ModelOption, mode: AudioMode): boolean {
|
|
32
|
+
if (mode === 'voice_design') return false
|
|
33
|
+
if (/^stable-audio-/i.test(entry.alias)) return mode === 'music' || mode === 'sfx'
|
|
34
|
+
if (entry.category === undefined) return true
|
|
35
|
+
if (entry.category === mode) return true
|
|
36
|
+
return entry.category === 'tts' && mode === 'tts'
|
|
37
|
+
}
|
|
38
|
+
|
|
24
39
|
export interface StudioReuse {
|
|
25
40
|
nonce: number
|
|
26
41
|
mode: AudioMode
|
|
@@ -209,7 +224,13 @@ export function StudioView(props: {
|
|
|
209
224
|
})
|
|
210
225
|
|
|
211
226
|
const [mode, setMode] = useState<AudioMode>('tts')
|
|
212
|
-
|
|
227
|
+
/** 每个模式独立的输入内容(TTS 文本 / 音乐·音效提示词 / 音色描述),切模式互不干扰。 */
|
|
228
|
+
const [promptByMode, setPromptByMode] = useState<Record<AudioMode, string>>({ tts: '', music: '', sfx: '', voice_design: '' })
|
|
229
|
+
const prompt = promptByMode[mode] ?? ''
|
|
230
|
+
const setPrompt = (next: string): void => {
|
|
231
|
+
const target = mode
|
|
232
|
+
setPromptByMode(current => (current[target] === next ? current : { ...current, [target]: next }))
|
|
233
|
+
}
|
|
213
234
|
const [previewText, setPreviewText] = useState('')
|
|
214
235
|
const [model, setModel] = useState('')
|
|
215
236
|
const [voice, setVoice] = useState('')
|
|
@@ -233,6 +254,8 @@ export function StudioView(props: {
|
|
|
233
254
|
// 提示词增强
|
|
234
255
|
const [enhancing, setEnhancing] = useState(false)
|
|
235
256
|
const [enhancePreview, setEnhancePreview] = useState<string | null>(null)
|
|
257
|
+
// 切换模式后增强预览属于旧模式,清除以免误用(prompt 本身按模式独立保留)。
|
|
258
|
+
useEffect(() => { setEnhancePreview(null) }, [mode])
|
|
236
259
|
// Stable Audio 参数(仅 Stability 渠道显示)
|
|
237
260
|
const [seed, setSeed] = useState('')
|
|
238
261
|
const [steps, setSteps] = useState('')
|
|
@@ -273,7 +296,7 @@ export function StudioView(props: {
|
|
|
273
296
|
const visibleModels = useMemo(() => {
|
|
274
297
|
if (mode === 'voice_design') return []
|
|
275
298
|
return modelOptions.models
|
|
276
|
-
.filter(entry => entry
|
|
299
|
+
.filter(entry => modelSuitableForMode(entry, mode))
|
|
277
300
|
.map(entry => entry.alias)
|
|
278
301
|
}, [modelOptions.models, mode])
|
|
279
302
|
|
|
@@ -528,7 +551,10 @@ export function StudioView(props: {
|
|
|
528
551
|
const bool = (key: string): boolean | undefined => typeof params[key] === 'boolean' ? params[key] as boolean : undefined
|
|
529
552
|
setMode(modeValue)
|
|
530
553
|
const promptValue = str('prompt')
|
|
531
|
-
if (promptValue !== undefined)
|
|
554
|
+
if (promptValue !== undefined) {
|
|
555
|
+
// 直接写入被恢复模式自己的槽位(此刻闭包 mode 仍是旧值)。
|
|
556
|
+
setPromptByMode(current => (current[modeValue] === promptValue ? current : { ...current, [modeValue]: promptValue }))
|
|
557
|
+
}
|
|
532
558
|
const modelValue = str('model') ?? singleModel
|
|
533
559
|
if (modelValue !== '') setModel(modelValue)
|
|
534
560
|
if (compareModelsRestore !== undefined && compareModelsRestore.length > 0) {
|
|
@@ -878,6 +904,8 @@ export function StudioView(props: {
|
|
|
878
904
|
|
|
879
905
|
const needModel = mode !== 'voice_design'
|
|
880
906
|
const runningCount = tasks.filter(task => task.status === 'running').length
|
|
907
|
+
/** 结果列只显示当前模式的任务(历史面板已按模式分组)。 */
|
|
908
|
+
const visibleTasks = useMemo(() => tasks.filter(task => task.mode === mode), [tasks, mode])
|
|
881
909
|
|
|
882
910
|
return (
|
|
883
911
|
<div className={css.studio}>
|
|
@@ -1031,7 +1059,11 @@ export function StudioView(props: {
|
|
|
1031
1059
|
{visibleModels.length === 0 ? <option value="">(当前模式暂无可用模型)</option> : null}
|
|
1032
1060
|
{groupedModels.map(group => (
|
|
1033
1061
|
<optgroup key={group.channelId} label={group.channelName}>
|
|
1034
|
-
{group.models.map(item =>
|
|
1062
|
+
{group.models.map(item => (
|
|
1063
|
+
<option key={item.alias} value={item.alias}>
|
|
1064
|
+
{/^stable-audio-/i.test(item.alias) ? `${item.alias} · 音乐/音效` : item.alias}
|
|
1065
|
+
</option>
|
|
1066
|
+
))}
|
|
1035
1067
|
</optgroup>
|
|
1036
1068
|
))}
|
|
1037
1069
|
</select>
|
|
@@ -1074,7 +1106,7 @@ export function StudioView(props: {
|
|
|
1074
1106
|
|
|
1075
1107
|
<div className={css.resultCol}>
|
|
1076
1108
|
{error !== null ? <p className={css.error}>{error}</p> : null}
|
|
1077
|
-
{
|
|
1109
|
+
{visibleTasks.length === 0 ? (
|
|
1078
1110
|
<div className={css.resultEmpty}>
|
|
1079
1111
|
<span className={css.resultEmptyIcon}>🎵</span>
|
|
1080
1112
|
<p>{tt('result.empty')}</p>
|
|
@@ -1091,7 +1123,7 @@ export function StudioView(props: {
|
|
|
1091
1123
|
</div>
|
|
1092
1124
|
) : (
|
|
1093
1125
|
<div className={css.taskList}>
|
|
1094
|
-
{
|
|
1126
|
+
{visibleTasks.map(task => {
|
|
1095
1127
|
const elapsed = task.finishedAt !== undefined
|
|
1096
1128
|
? Math.round((task.finishedAt - task.startedAt) / 1000)
|
|
1097
1129
|
: Math.round((Date.now() - task.startedAt) / 1000)
|
package/src/index.ts
CHANGED
|
@@ -16,7 +16,7 @@ import z from 'schemastery'
|
|
|
16
16
|
import type {} from '@deepseek-ai/dsh-host-webserver'
|
|
17
17
|
import type {} from '@deepseek-ai/dsh-system-prompt'
|
|
18
18
|
import type {} from '@deepseek-ai/dsh-tools'
|
|
19
|
-
import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type ModelMapping } from './protocol.ts'
|
|
19
|
+
import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type LlmModelOption, type ModelMapping } from './protocol.ts'
|
|
20
20
|
import { createGenerationBudget } from './audio-scheduler.ts'
|
|
21
21
|
import { enhancePromptText } from './prompt-enhance.ts'
|
|
22
22
|
import { AudioGenError } from './audio-engine.ts'
|
|
@@ -45,6 +45,11 @@ export interface Config {
|
|
|
45
45
|
autoSaveToLibrary?: boolean
|
|
46
46
|
/** 设置卡按文本编辑,保存值可能是数字或数字字符串。 */
|
|
47
47
|
maxConcurrentGenerations?: number | string
|
|
48
|
+
/**
|
|
49
|
+
* 提示词增强模型,格式 "provider|model"(如 "deepseek-official|deepseek-v4-flash-vision-exp");
|
|
50
|
+
* 空串表示跟随 Agent 默认模型(「设置 → 模型」)。
|
|
51
|
+
*/
|
|
52
|
+
enhanceModel?: string
|
|
48
53
|
}
|
|
49
54
|
|
|
50
55
|
const DEFAULT_MAX_CONCURRENT = 5
|
|
@@ -68,6 +73,7 @@ export const Config: z<Config> = z.object({
|
|
|
68
73
|
defaultModel: z.string().default(''),
|
|
69
74
|
autoSaveToLibrary: z.boolean().default(false),
|
|
70
75
|
maxConcurrentGenerations: z.union([z.number(), z.string()]).default(DEFAULT_MAX_CONCURRENT),
|
|
76
|
+
enhanceModel: z.string().default(''),
|
|
71
77
|
})
|
|
72
78
|
|
|
73
79
|
const DEFAULT_ENABLED = true
|
|
@@ -158,6 +164,8 @@ export interface EffectiveConfig {
|
|
|
158
164
|
defaultChannelId: string
|
|
159
165
|
defaultModel: string
|
|
160
166
|
autoSaveToLibrary: boolean
|
|
167
|
+
/** 提示词增强模型("provider|model";空串 = 跟随 Agent 默认模型)。 */
|
|
168
|
+
enhanceModel: string
|
|
161
169
|
maxConcurrentGenerations: number
|
|
162
170
|
}
|
|
163
171
|
|
|
@@ -186,6 +194,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
186
194
|
})),
|
|
187
195
|
defaultChannelId,
|
|
188
196
|
defaultModel: typeof value.defaultModel === 'string' ? value.defaultModel.trim() : '',
|
|
197
|
+
enhanceModel: typeof value.enhanceModel === 'string' ? value.enhanceModel.trim() : '',
|
|
189
198
|
autoSaveToLibrary: value.autoSaveToLibrary === true,
|
|
190
199
|
maxConcurrentGenerations: (() => {
|
|
191
200
|
const rawMax = value.maxConcurrentGenerations
|
|
@@ -198,11 +207,54 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
198
207
|
// 全局并发闸门:所有上游调用(面板路由 + Agent 工具)共享「最大并发生成数」。
|
|
199
208
|
const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations)
|
|
200
209
|
|
|
201
|
-
//
|
|
210
|
+
// 提示词增强:优先用设置的增强模型(enhanceModel),否则复用 Agent 默认模型
|
|
211
|
+
// (agent-default-model 设置);面板与 Agent 工具共用。
|
|
202
212
|
const enhance = async (prompt: string, mode: AudioMode): Promise<string> => {
|
|
203
213
|
const seam = ctx.get('settings') as unknown as SettingsSeam
|
|
204
214
|
if (seam?.describe === undefined) throw new AudioGenError('设置服务不可用,无法增强提示词', 'settings-unavailable')
|
|
205
|
-
return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode)
|
|
215
|
+
return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode, enhanceSelectionOf(resolve()))
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/** 解析设置的增强模型("provider|model");空/非法值返回 undefined = 跟随默认。 */
|
|
219
|
+
const enhanceSelectionOf = (config: EffectiveConfig): { provider: string; model: string } | undefined => {
|
|
220
|
+
const raw = config.enhanceModel ?? ''
|
|
221
|
+
const sep = raw.indexOf('|')
|
|
222
|
+
if (sep <= 0 || sep >= raw.length - 1) return undefined
|
|
223
|
+
const provider = raw.slice(0, sep).trim()
|
|
224
|
+
const model = raw.slice(sep + 1).trim()
|
|
225
|
+
return provider !== '' && model !== '' ? { provider, model } : undefined
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/** 读取「设置 → 模型」目录:各提供方 + 可广播的模型列表(增强模型下拉候选)。 */
|
|
229
|
+
const llmModelOptions = async (): Promise<LlmModelOption[]> => {
|
|
230
|
+
const llm = ctx.get('llm') as {
|
|
231
|
+
listProviders?: () => Array<{ id?: string; name?: string }>
|
|
232
|
+
listConfigurableProviders?: () => Array<{ provider?: string; displayName?: string }>
|
|
233
|
+
listModels?: (provider: string) => Promise<Array<{ id?: string; name?: string }>>
|
|
234
|
+
} | undefined
|
|
235
|
+
if (llm === undefined || llm.listProviders === undefined || llm.listModels === undefined) return []
|
|
236
|
+
const options: LlmModelOption[] = []
|
|
237
|
+
const directory = new Map<string, string>()
|
|
238
|
+
if (llm.listConfigurableProviders !== undefined) {
|
|
239
|
+
for (const entry of llm.listConfigurableProviders() ?? []) {
|
|
240
|
+
if (typeof entry.provider === 'string' && entry.provider !== '' && typeof entry.displayName === 'string') {
|
|
241
|
+
directory.set(entry.provider, entry.displayName)
|
|
242
|
+
}
|
|
243
|
+
}
|
|
244
|
+
}
|
|
245
|
+
for (const info of llm.listProviders() ?? []) {
|
|
246
|
+
const provider = typeof info.id === 'string' ? info.id : ''
|
|
247
|
+
if (provider === '') continue
|
|
248
|
+
let models: Array<{ id?: string; name?: string }> = []
|
|
249
|
+
try { models = (await llm.listModels(provider)) ?? [] } catch { models = [] }
|
|
250
|
+
for (const model of models) {
|
|
251
|
+
const id = typeof model.id === 'string' ? model.id.trim() : ''
|
|
252
|
+
if (id === '') continue
|
|
253
|
+
const name = typeof model.name === 'string' && model.name.trim() !== '' ? model.name.trim() : id
|
|
254
|
+
options.push({ provider, providerName: directory.get(provider) ?? provider, id, name })
|
|
255
|
+
}
|
|
256
|
+
}
|
|
257
|
+
return options
|
|
206
258
|
}
|
|
207
259
|
|
|
208
260
|
const channelsView = (): ChannelsView => {
|
|
@@ -219,6 +271,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
219
271
|
autoSave: () => resolve().autoSaveToLibrary,
|
|
220
272
|
budget,
|
|
221
273
|
enhance,
|
|
274
|
+
llmModelOptions,
|
|
222
275
|
})
|
|
223
276
|
const disposers = routes.map(route => ctx.webServer.register(route))
|
|
224
277
|
return () => { for (const dispose of disposers) dispose() }
|