dsh-audiogen 0.4.4 → 0.4.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +438 -340
- package/lib/client.js.map +1 -1
- package/lib/index.js +111 -2
- package/package.json +1 -1
- package/src/agent-audio-tools.ts +11 -0
- package/src/client/api.ts +8 -1
- package/src/client/audio-panel.module.css +104 -14
- package/src/client/studio-view.tsx +45 -0
- package/src/index.ts +12 -1
- package/src/prompt-enhance.ts +89 -0
- package/src/protocol.ts +4 -1
- package/src/routes.ts +24 -1
package/lib/index.js
CHANGED
|
@@ -24,6 +24,8 @@ const SETTINGS_API = {
|
|
|
24
24
|
const GENERATE_API = "/api/dsh-audiogen/generate";
|
|
25
25
|
/** Loopback-only task cancellation route (aborts the host-side upstream call). */
|
|
26
26
|
const TASK_API = { cancel: "/api/dsh-audiogen/task/cancel" };
|
|
27
|
+
/** Loopback-only prompt enhancement route (uses the agent's default model). */
|
|
28
|
+
const ENHANCE_API = "/api/dsh-audiogen/prompt/enhance";
|
|
27
29
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
28
30
|
const PRESETS_API = "/api/dsh-audiogen/presets";
|
|
29
31
|
/** Host-mediated model/voice discovery endpoint. */
|
|
@@ -864,6 +866,66 @@ async function generateAudio(channel, request, signal) {
|
|
|
864
866
|
return genericAudio(channel, request, signal);
|
|
865
867
|
}
|
|
866
868
|
//#endregion
|
|
869
|
+
//#region src/prompt-enhance.ts
|
|
870
|
+
/** 按生成模式给出增强指令(系统提示)。 */
|
|
871
|
+
function instructionsFor(mode) {
|
|
872
|
+
return [
|
|
873
|
+
"你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,",
|
|
874
|
+
"请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。",
|
|
875
|
+
"只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。",
|
|
876
|
+
"保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。",
|
|
877
|
+
"描述控制在 200-600 字左右。"
|
|
878
|
+
].join("") + {
|
|
879
|
+
tts: "这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。",
|
|
880
|
+
music: "这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。",
|
|
881
|
+
sfx: "这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。",
|
|
882
|
+
voice_design: "这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。"
|
|
883
|
+
}[mode];
|
|
884
|
+
}
|
|
885
|
+
/** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
|
|
886
|
+
async function enhancePromptText(deps, prompt, mode) {
|
|
887
|
+
const text = prompt.trim();
|
|
888
|
+
if (text === "") throw new AudioGenError("提示词为空,无法增强", "enhance-empty-prompt");
|
|
889
|
+
const value = (deps.settings.describe({ redactSecrets: true }) ?? []).find((candidate) => String(candidate.ns) === "agent-default-model")?.value ?? {};
|
|
890
|
+
const provider = typeof value.provider === "string" && value.provider.trim() !== "" ? value.provider.trim() : "";
|
|
891
|
+
const model = typeof value.model === "string" && value.model.trim() !== "" ? value.model.trim() : "";
|
|
892
|
+
if (provider === "" || model === "") throw new AudioGenError("未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型", "no-default-model");
|
|
893
|
+
const runtime = deps.llm?.();
|
|
894
|
+
if (runtime === void 0 || runtime.stream === void 0) throw new AudioGenError("宿主 LLM 服务不可用(ctx.llm 未注册)", "llm-unavailable");
|
|
895
|
+
const controller = new AbortController();
|
|
896
|
+
const timer = setTimeout(() => controller.abort(new DOMException("The operation timed out.", "TimeoutError")), 3e4);
|
|
897
|
+
timer.unref?.();
|
|
898
|
+
let output = "";
|
|
899
|
+
try {
|
|
900
|
+
for await (const chunk of runtime.stream({
|
|
901
|
+
provider,
|
|
902
|
+
model,
|
|
903
|
+
messages: [{
|
|
904
|
+
role: "user",
|
|
905
|
+
content: text
|
|
906
|
+
}],
|
|
907
|
+
system: instructionsFor(mode),
|
|
908
|
+
temperature: .7,
|
|
909
|
+
maxTokens: 1200,
|
|
910
|
+
signal: controller.signal
|
|
911
|
+
})) {
|
|
912
|
+
const record = chunk;
|
|
913
|
+
if (record.type === "text-delta" && typeof record.text === "string") output += record.text;
|
|
914
|
+
else if (record.type === "block-end" && record.block !== void 0 && record.block.type === "text" && typeof record.block.text === "string") output += record.block.text;
|
|
915
|
+
}
|
|
916
|
+
} finally {
|
|
917
|
+
clearTimeout(timer);
|
|
918
|
+
}
|
|
919
|
+
const result = stripFences(output.trim());
|
|
920
|
+
if (result === "") throw new AudioGenError("模型未返回增强内容:请检查「设置 → 模型」的默认模型是否可用(或稍后重试)", "enhance-empty-result");
|
|
921
|
+
return result;
|
|
922
|
+
}
|
|
923
|
+
/** 去掉模型可能包裹的 ``` 代码围栏。 */
|
|
924
|
+
function stripFences(value) {
|
|
925
|
+
if (value === "") return value;
|
|
926
|
+
return value.replace(/^```[a-zA-Z]*\s*\n?/, "").replace(/\n?```\s*$/, "").trim();
|
|
927
|
+
}
|
|
928
|
+
//#endregion
|
|
867
929
|
//#region src/audio-presets.ts
|
|
868
930
|
const AUDIO_PRESETS = [
|
|
869
931
|
{
|
|
@@ -2046,6 +2108,36 @@ function makeRoutes(deps) {
|
|
|
2046
2108
|
});
|
|
2047
2109
|
}
|
|
2048
2110
|
},
|
|
2111
|
+
{
|
|
2112
|
+
kind: "exact",
|
|
2113
|
+
path: ENHANCE_API,
|
|
2114
|
+
handler: async (req, res) => {
|
|
2115
|
+
if (!guard(req, res, "POST")) return;
|
|
2116
|
+
const body = await readJsonBody(req);
|
|
2117
|
+
const prompt = typeof body?.prompt === "string" ? body.prompt.trim() : "";
|
|
2118
|
+
if (prompt === "") {
|
|
2119
|
+
writeJson(res, 200, {
|
|
2120
|
+
ok: false,
|
|
2121
|
+
code: "bad-request",
|
|
2122
|
+
message: "prompt is required"
|
|
2123
|
+
});
|
|
2124
|
+
return;
|
|
2125
|
+
}
|
|
2126
|
+
const mode = body?.mode === "music" ? "music" : body?.mode === "sfx" ? "sfx" : body?.mode === "voice_design" ? "voice_design" : "tts";
|
|
2127
|
+
try {
|
|
2128
|
+
writeJson(res, 200, {
|
|
2129
|
+
ok: true,
|
|
2130
|
+
enhanced: await deps.enhance(prompt, mode)
|
|
2131
|
+
});
|
|
2132
|
+
} catch (error) {
|
|
2133
|
+
writeJson(res, 200, {
|
|
2134
|
+
ok: false,
|
|
2135
|
+
code: "enhance-failed",
|
|
2136
|
+
message: messageOf(error)
|
|
2137
|
+
});
|
|
2138
|
+
}
|
|
2139
|
+
}
|
|
2140
|
+
},
|
|
2049
2141
|
{
|
|
2050
2142
|
kind: "prefix",
|
|
2051
2143
|
path: AUDIO_API.file,
|
|
@@ -2483,6 +2575,10 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
2483
2575
|
type: "string",
|
|
2484
2576
|
description: "Optional preview text for voice_design."
|
|
2485
2577
|
},
|
|
2578
|
+
enhance_prompt: {
|
|
2579
|
+
type: "boolean",
|
|
2580
|
+
description: "Enhance the prompt with the agent default model before generating (uses the configured model settings, no extra key). Best-effort: on failure the original prompt is used."
|
|
2581
|
+
},
|
|
2486
2582
|
speed: {
|
|
2487
2583
|
type: "number",
|
|
2488
2584
|
description: "Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1)."
|
|
@@ -2725,6 +2821,9 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
2725
2821
|
/** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
|
|
2726
2822
|
const runOne = async (picked) => {
|
|
2727
2823
|
const request = buildRequest(picked);
|
|
2824
|
+
if (args.enhance_prompt === true && config.enhance !== void 0) try {
|
|
2825
|
+
request.prompt = await config.enhance(request.prompt, request.mode);
|
|
2826
|
+
} catch {}
|
|
2728
2827
|
try {
|
|
2729
2828
|
const release = await (config.budget?.acquire(exec.signal) ?? Promise.resolve(() => {}));
|
|
2730
2829
|
let outputs;
|
|
@@ -3137,6 +3236,14 @@ function apply(ctx, config) {
|
|
|
3137
3236
|
};
|
|
3138
3237
|
};
|
|
3139
3238
|
const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations);
|
|
3239
|
+
const enhance = async (prompt, mode) => {
|
|
3240
|
+
const seam = ctx.get("settings");
|
|
3241
|
+
if (seam?.describe === void 0) throw new AudioGenError("设置服务不可用,无法增强提示词", "settings-unavailable");
|
|
3242
|
+
return enhancePromptText({
|
|
3243
|
+
settings: seam,
|
|
3244
|
+
llm: () => ctx.get("llm")
|
|
3245
|
+
}, prompt, mode);
|
|
3246
|
+
};
|
|
3140
3247
|
const channelsView = () => {
|
|
3141
3248
|
const value = resolve();
|
|
3142
3249
|
return {
|
|
@@ -3151,7 +3258,8 @@ function apply(ctx, config) {
|
|
|
3151
3258
|
settings: seam,
|
|
3152
3259
|
resolveChannels: channelsView,
|
|
3153
3260
|
autoSave: () => resolve().autoSaveToLibrary,
|
|
3154
|
-
budget
|
|
3261
|
+
budget,
|
|
3262
|
+
enhance
|
|
3155
3263
|
}).map((route) => ctx.webServer.register(route));
|
|
3156
3264
|
return () => {
|
|
3157
3265
|
for (const dispose of disposers) dispose();
|
|
@@ -3167,7 +3275,8 @@ function apply(ctx, config) {
|
|
|
3167
3275
|
channels: value.channels,
|
|
3168
3276
|
defaultChannelId: value.defaultChannelId,
|
|
3169
3277
|
autoSaveToLibrary: value.autoSaveToLibrary,
|
|
3170
|
-
budget
|
|
3278
|
+
budget,
|
|
3279
|
+
enhance
|
|
3171
3280
|
};
|
|
3172
3281
|
}), "dsh-audiogen: agent audio tools");
|
|
3173
3282
|
});
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-audiogen",
|
|
3
3
|
"description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.6",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"exports": {
|
package/src/agent-audio-tools.ts
CHANGED
|
@@ -21,6 +21,8 @@ export interface AgentAudioToolConfig {
|
|
|
21
21
|
autoSaveToLibrary: boolean
|
|
22
22
|
/** 全局并发闸门(与面板路由共享「最大并发生成数」)。 */
|
|
23
23
|
budget?: GenerationBudget
|
|
24
|
+
/** 提示词增强(复用 Agent 默认模型);enhance_prompt=true 时在生成前调用。 */
|
|
25
|
+
enhance?: (prompt: string, mode: AudioMode) => Promise<string>
|
|
24
26
|
}
|
|
25
27
|
|
|
26
28
|
interface AgentAudioRef {
|
|
@@ -157,6 +159,7 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
157
159
|
},
|
|
158
160
|
voice: { type: 'string', description: 'Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio.' },
|
|
159
161
|
preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
|
|
162
|
+
enhance_prompt: { type: 'boolean', description: 'Enhance the prompt with the agent default model before generating (uses the configured model settings, no extra key). Best-effort: on failure the original prompt is used.' },
|
|
160
163
|
speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1).' },
|
|
161
164
|
duration: { type: 'number', description: 'Requested duration in seconds for music/sfx.' },
|
|
162
165
|
lyrics: { type: 'string', description: 'Lyrics for music generation (MiniMax music-3.0/music-cover). Required unless is_instrumental is true. Split verses with an empty line.' },
|
|
@@ -302,6 +305,14 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
|
|
|
302
305
|
/** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
|
|
303
306
|
const runOne = async (picked: { channel: AudioChannel; alias: string; upstream: string }): Promise<AgentAudioGroup> => {
|
|
304
307
|
const request = buildRequest(picked)
|
|
308
|
+
// 可选:生成前用 Agent 默认模型增强 prompt(失败则沿用原文)
|
|
309
|
+
if (args.enhance_prompt === true && config.enhance !== undefined) {
|
|
310
|
+
try {
|
|
311
|
+
request.prompt = await config.enhance(request.prompt, request.mode)
|
|
312
|
+
} catch {
|
|
313
|
+
// 增强失败不阻断生成
|
|
314
|
+
}
|
|
315
|
+
}
|
|
305
316
|
try {
|
|
306
317
|
// 与面板路由共享全局并发闸门(限流时排队;取消时立即出队)。
|
|
307
318
|
const release = await (config.budget?.acquire(exec.signal) ?? Promise.resolve(() => { /* 默认不限制 */ }))
|
package/src/client/api.ts
CHANGED
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
*/
|
|
5
5
|
|
|
6
6
|
import {
|
|
7
|
-
GENERATE_API, HISTORY_API, LIBRARY_API, TASK_API,
|
|
7
|
+
ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, TASK_API,
|
|
8
8
|
type GenerateAudioRequest, type GeneratedAudio, type HistoryEntry,
|
|
9
9
|
type LibraryEntry, type LibrarySaveRequest, type LibraryUpdateRequest,
|
|
10
10
|
} from '../protocol.ts'
|
|
@@ -42,6 +42,13 @@ export class AudiogenApi {
|
|
|
42
42
|
await postJson(TASK_API.cancel, { taskId }).catch(() => { /* best-effort */ })
|
|
43
43
|
}
|
|
44
44
|
|
|
45
|
+
/** 提示词增强(复用 Agent 默认模型)。 */
|
|
46
|
+
async enhancePrompt(prompt: string, mode: string): Promise<{ ok: boolean; enhanced?: string; code?: string; message?: string }> {
|
|
47
|
+
const response = await postJson(ENHANCE_API, { prompt, mode })
|
|
48
|
+
const body = await response.json() as { ok?: boolean; enhanced?: string; code?: string; message?: string }
|
|
49
|
+
return { ok: body.ok === true, ...(body.enhanced === undefined ? {} : { enhanced: body.enhanced }), ...(body.code === undefined ? {} : { code: body.code }), ...(body.message === undefined ? {} : { message: body.message }) }
|
|
50
|
+
}
|
|
51
|
+
|
|
45
52
|
async history(): Promise<HistoryEntry[]> {
|
|
46
53
|
const response = await postJson(HISTORY_API.list, {})
|
|
47
54
|
const body = await response.json() as { ok?: boolean; history?: HistoryEntry[] }
|
|
@@ -97,9 +97,9 @@
|
|
|
97
97
|
flex-direction: column;
|
|
98
98
|
gap: 11px;
|
|
99
99
|
flex: none;
|
|
100
|
-
width:
|
|
101
|
-
min-width:
|
|
102
|
-
max-width:
|
|
100
|
+
width: 380px;
|
|
101
|
+
min-width: 320px;
|
|
102
|
+
max-width: 440px;
|
|
103
103
|
min-height: 0;
|
|
104
104
|
overflow-y: auto;
|
|
105
105
|
padding: 2px;
|
|
@@ -111,8 +111,8 @@
|
|
|
111
111
|
.formCol::-webkit-scrollbar-thumb { background: var(--dsw-alias-border-l2); border-radius: 999px; }
|
|
112
112
|
|
|
113
113
|
.modeRow {
|
|
114
|
-
display:
|
|
115
|
-
|
|
114
|
+
display: flex;
|
|
115
|
+
flex-wrap: wrap;
|
|
116
116
|
gap: 6px;
|
|
117
117
|
}
|
|
118
118
|
|
|
@@ -120,8 +120,10 @@
|
|
|
120
120
|
display: flex;
|
|
121
121
|
align-items: center;
|
|
122
122
|
justify-content: center;
|
|
123
|
+
flex: 1 1 auto;
|
|
124
|
+
min-width: 0;
|
|
123
125
|
min-height: 34px;
|
|
124
|
-
padding: 6px
|
|
126
|
+
padding: 6px 10px;
|
|
125
127
|
border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
|
|
126
128
|
border-radius: 9px;
|
|
127
129
|
background: var(--dsw-alias-bg-layer-1, #fff);
|
|
@@ -258,6 +260,7 @@
|
|
|
258
260
|
.resultCol {
|
|
259
261
|
flex: 1;
|
|
260
262
|
min-width: 0;
|
|
263
|
+
max-width: 760px;
|
|
261
264
|
display: flex;
|
|
262
265
|
flex-direction: column;
|
|
263
266
|
gap: 10px;
|
|
@@ -960,7 +963,7 @@
|
|
|
960
963
|
|
|
961
964
|
.resultGroupError {
|
|
962
965
|
font-size: 12px;
|
|
963
|
-
color: #dc2626;
|
|
966
|
+
color: var(--dsw-alias-state-error-primary, #dc2626);
|
|
964
967
|
}
|
|
965
968
|
|
|
966
969
|
.resultGroupCount {
|
|
@@ -1021,7 +1024,7 @@
|
|
|
1021
1024
|
}
|
|
1022
1025
|
|
|
1023
1026
|
.taskCard[data-state='failed'] {
|
|
1024
|
-
border-color: #dc2626;
|
|
1027
|
+
border-color: var(--dsw-alias-state-error-primary, #dc2626);
|
|
1025
1028
|
}
|
|
1026
1029
|
|
|
1027
1030
|
.taskCard[data-state='cancelled'] {
|
|
@@ -1051,11 +1054,11 @@
|
|
|
1051
1054
|
}
|
|
1052
1055
|
|
|
1053
1056
|
.taskStatus[data-state='done'] {
|
|
1054
|
-
color: #16a34a;
|
|
1057
|
+
color: var(--dsw-alias-state-success-primary, #16a34a);
|
|
1055
1058
|
}
|
|
1056
1059
|
|
|
1057
1060
|
.taskStatus[data-state='failed'] {
|
|
1058
|
-
color: #dc2626;
|
|
1061
|
+
color: var(--dsw-alias-state-error-primary, #dc2626);
|
|
1059
1062
|
}
|
|
1060
1063
|
|
|
1061
1064
|
.taskActions {
|
|
@@ -1107,11 +1110,11 @@
|
|
|
1107
1110
|
.historyCompareBadge {
|
|
1108
1111
|
font-size: 11px;
|
|
1109
1112
|
font-weight: 600;
|
|
1110
|
-
color: #
|
|
1111
|
-
border: 1px solid #
|
|
1113
|
+
color: var(--dsw-alias-label-secondary, #6b7280);
|
|
1114
|
+
border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
|
|
1112
1115
|
border-radius: 999px;
|
|
1113
1116
|
padding: 1px 8px;
|
|
1114
|
-
background: #
|
|
1117
|
+
background: var(--dsw-alias-bg-layer-3, #fff);
|
|
1115
1118
|
white-space: nowrap;
|
|
1116
1119
|
}
|
|
1117
1120
|
|
|
@@ -1221,7 +1224,7 @@
|
|
|
1221
1224
|
}
|
|
1222
1225
|
|
|
1223
1226
|
.historyIcon[title^='删除']:hover {
|
|
1224
|
-
color: #dc2626;
|
|
1227
|
+
color: var(--dsw-alias-state-error-primary, #dc2626);
|
|
1225
1228
|
}
|
|
1226
1229
|
|
|
1227
1230
|
/* 历史 prompt 两行截断 */
|
|
@@ -1231,3 +1234,90 @@
|
|
|
1231
1234
|
-webkit-box-orient: vertical;
|
|
1232
1235
|
overflow: hidden;
|
|
1233
1236
|
}
|
|
1237
|
+
|
|
1238
|
+
/* ------------------------------------------------------- P3 打磨:响应式与统一质感 */
|
|
1239
|
+
|
|
1240
|
+
/* 焦点态统一(键盘可达性) */
|
|
1241
|
+
.input:focus-visible,
|
|
1242
|
+
.select:focus-visible,
|
|
1243
|
+
.textarea:focus-visible,
|
|
1244
|
+
.modeButton:focus-visible,
|
|
1245
|
+
.ghostButton:focus-visible,
|
|
1246
|
+
.historyAction:focus-visible,
|
|
1247
|
+
.historyIcon:focus-visible {
|
|
1248
|
+
outline: 2px solid var(--dsw-alias-interactive-bg-hover-accent, rgba(38, 49, 72, 0.35));
|
|
1249
|
+
outline-offset: 1px;
|
|
1250
|
+
}
|
|
1251
|
+
|
|
1252
|
+
.generate:focus-visible {
|
|
1253
|
+
outline: 2px solid var(--dsw-alias-interactive-bg-hover-accent, rgba(38, 49, 72, 0.35));
|
|
1254
|
+
outline-offset: 2px;
|
|
1255
|
+
}
|
|
1256
|
+
|
|
1257
|
+
/* 中窄屏:历史移到下方整行 */
|
|
1258
|
+
@media (max-width: 1100px) {
|
|
1259
|
+
.studio {
|
|
1260
|
+
flex-wrap: wrap;
|
|
1261
|
+
}
|
|
1262
|
+
|
|
1263
|
+
.historyCol {
|
|
1264
|
+
width: 100%;
|
|
1265
|
+
min-width: 0;
|
|
1266
|
+
max-width: none;
|
|
1267
|
+
}
|
|
1268
|
+
}
|
|
1269
|
+
|
|
1270
|
+
/* 窄屏:单列堆叠 */
|
|
1271
|
+
@media (max-width: 720px) {
|
|
1272
|
+
.formCol {
|
|
1273
|
+
width: 100%;
|
|
1274
|
+
min-width: 0;
|
|
1275
|
+
max-width: none;
|
|
1276
|
+
}
|
|
1277
|
+
|
|
1278
|
+
.resultCol {
|
|
1279
|
+
min-height: 320px;
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
1282
|
+
|
|
1283
|
+
/* 提示词增强 */
|
|
1284
|
+
.formSectionRow {
|
|
1285
|
+
display: flex;
|
|
1286
|
+
align-items: flex-end;
|
|
1287
|
+
justify-content: space-between;
|
|
1288
|
+
gap: 8px;
|
|
1289
|
+
}
|
|
1290
|
+
|
|
1291
|
+
.formSectionRow .formSection {
|
|
1292
|
+
flex: 1;
|
|
1293
|
+
}
|
|
1294
|
+
|
|
1295
|
+
.enhanceCard {
|
|
1296
|
+
border: 1px solid var(--dsw-alias-border-l1, #e5e7eb);
|
|
1297
|
+
border-radius: 10px;
|
|
1298
|
+
padding: 10px;
|
|
1299
|
+
background: var(--dsw-alias-bg-layer-2, #fafafa);
|
|
1300
|
+
display: flex;
|
|
1301
|
+
flex-direction: column;
|
|
1302
|
+
gap: 8px;
|
|
1303
|
+
}
|
|
1304
|
+
|
|
1305
|
+
.enhanceCardHead {
|
|
1306
|
+
display: flex;
|
|
1307
|
+
align-items: center;
|
|
1308
|
+
justify-content: space-between;
|
|
1309
|
+
gap: 8px;
|
|
1310
|
+
font-size: 12px;
|
|
1311
|
+
color: var(--dsw-alias-label-secondary, #6b7280);
|
|
1312
|
+
}
|
|
1313
|
+
|
|
1314
|
+
.enhanceActions {
|
|
1315
|
+
display: flex;
|
|
1316
|
+
gap: 6px;
|
|
1317
|
+
}
|
|
1318
|
+
|
|
1319
|
+
.enhanceActionsRow {
|
|
1320
|
+
display: flex;
|
|
1321
|
+
justify-content: flex-end;
|
|
1322
|
+
margin-top: -4px;
|
|
1323
|
+
}
|
|
@@ -230,6 +230,9 @@ export function StudioView(props: {
|
|
|
230
230
|
const [bitrate, setBitrate] = useState('')
|
|
231
231
|
const [audioChannel, setAudioChannel] = useState('')
|
|
232
232
|
const [subtitle, setSubtitle] = useState(false)
|
|
233
|
+
// 提示词增强
|
|
234
|
+
const [enhancing, setEnhancing] = useState(false)
|
|
235
|
+
const [enhancePreview, setEnhancePreview] = useState<string | null>(null)
|
|
233
236
|
// Stable Audio 参数(仅 Stability 渠道显示)
|
|
234
237
|
const [seed, setSeed] = useState('')
|
|
235
238
|
const [steps, setSteps] = useState('')
|
|
@@ -524,6 +527,8 @@ export function StudioView(props: {
|
|
|
524
527
|
}
|
|
525
528
|
const bool = (key: string): boolean | undefined => typeof params[key] === 'boolean' ? params[key] as boolean : undefined
|
|
526
529
|
setMode(modeValue)
|
|
530
|
+
const promptValue = str('prompt')
|
|
531
|
+
if (promptValue !== undefined) setPrompt(promptValue)
|
|
527
532
|
const modelValue = str('model') ?? singleModel
|
|
528
533
|
if (modelValue !== '') setModel(modelValue)
|
|
529
534
|
if (compareModelsRestore !== undefined && compareModelsRestore.length > 0) {
|
|
@@ -578,6 +583,28 @@ export function StudioView(props: {
|
|
|
578
583
|
props.showToast('已恢复该次生成的配置,可直接再次生成')
|
|
579
584
|
}
|
|
580
585
|
|
|
586
|
+
/** 调用宿主增强(Agent 默认模型),结果先预览再应用。 */
|
|
587
|
+
const runEnhance = async (): Promise<void> => {
|
|
588
|
+
if (prompt.trim() === '') {
|
|
589
|
+
setError('请先输入文本/提示词,再点击增强')
|
|
590
|
+
return
|
|
591
|
+
}
|
|
592
|
+
setEnhancing(true)
|
|
593
|
+
setError(null)
|
|
594
|
+
try {
|
|
595
|
+
const result = await api.enhancePrompt(prompt.trim(), mode)
|
|
596
|
+
if (result.ok !== true || result.enhanced === undefined || result.enhanced.trim() === '') {
|
|
597
|
+
setError(result.message ?? '增强失败,请稍后重试')
|
|
598
|
+
return
|
|
599
|
+
}
|
|
600
|
+
setEnhancePreview(result.enhanced.trim())
|
|
601
|
+
} catch (err) {
|
|
602
|
+
setError(err instanceof Error ? err.message : String(err))
|
|
603
|
+
} finally {
|
|
604
|
+
setEnhancing(false)
|
|
605
|
+
}
|
|
606
|
+
}
|
|
607
|
+
|
|
581
608
|
/** 删除历史记录(对比任务卡删除该任务的全部模型条目)。 */
|
|
582
609
|
const deleteHistoryEntries = async (ids: string[]): Promise<void> => {
|
|
583
610
|
try {
|
|
@@ -875,6 +902,24 @@ export function StudioView(props: {
|
|
|
875
902
|
<span>{mode === 'voice_design' ? '音色描述' : mode === 'tts' ? '文本' : '提示词'}</span>
|
|
876
903
|
<textarea className={css.textarea} value={prompt} onChange={event => setPrompt(event.target.value)} placeholder={tt('prompt.placeholder')} />
|
|
877
904
|
</label>
|
|
905
|
+
<div className={css.enhanceActionsRow}>
|
|
906
|
+
<button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>
|
|
907
|
+
{enhancing ? '增强中…' : '✨ 增强提示词'}
|
|
908
|
+
</button>
|
|
909
|
+
</div>
|
|
910
|
+
{enhancePreview !== null ? (
|
|
911
|
+
<div className={css.enhanceCard}>
|
|
912
|
+
<div className={css.enhanceCardHead}>
|
|
913
|
+
<strong>增强结果({modeLabelOf(mode)})</strong>
|
|
914
|
+
<span className={css.enhanceActions}>
|
|
915
|
+
<button type="button" className={css.ghostButton} onClick={() => { setPrompt(enhancePreview); setEnhancePreview(null); props.showToast('已应用增强结果') }}>应用</button>
|
|
916
|
+
<button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>重新生成</button>
|
|
917
|
+
<button type="button" className={css.ghostButton} onClick={() => setEnhancePreview(null)}>放弃</button>
|
|
918
|
+
</span>
|
|
919
|
+
</div>
|
|
920
|
+
<textarea className={css.textarea} value={enhancePreview} readOnly />
|
|
921
|
+
</div>
|
|
922
|
+
) : null}
|
|
878
923
|
|
|
879
924
|
{mode === 'voice_design' ? (
|
|
880
925
|
<>
|
package/src/index.ts
CHANGED
|
@@ -16,8 +16,10 @@ import z from 'schemastery'
|
|
|
16
16
|
import type {} from '@deepseek-ai/dsh-host-webserver'
|
|
17
17
|
import type {} from '@deepseek-ai/dsh-system-prompt'
|
|
18
18
|
import type {} from '@deepseek-ai/dsh-tools'
|
|
19
|
-
import { AUDIOGEN_SETTINGS_NAMESPACE, type ChannelConfig, type ModelMapping } from './protocol.ts'
|
|
19
|
+
import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type ModelMapping } from './protocol.ts'
|
|
20
20
|
import { createGenerationBudget } from './audio-scheduler.ts'
|
|
21
|
+
import { enhancePromptText } from './prompt-enhance.ts'
|
|
22
|
+
import { AudioGenError } from './audio-engine.ts'
|
|
21
23
|
import { makeRoutes, type ChannelsView, type SettingsSeam } from './routes.ts'
|
|
22
24
|
import type { AudioChannel } from './audio-engine.ts'
|
|
23
25
|
import { registerAgentAudioTools, type AgentAudioToolConfig } from './agent-audio-tools.ts'
|
|
@@ -196,6 +198,13 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
196
198
|
// 全局并发闸门:所有上游调用(面板路由 + Agent 工具)共享「最大并发生成数」。
|
|
197
199
|
const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations)
|
|
198
200
|
|
|
201
|
+
// 提示词增强:复用 Agent 默认模型(agent-default-model 设置),面板与 Agent 工具共用。
|
|
202
|
+
const enhance = async (prompt: string, mode: AudioMode): Promise<string> => {
|
|
203
|
+
const seam = ctx.get('settings') as unknown as SettingsSeam
|
|
204
|
+
if (seam?.describe === undefined) throw new AudioGenError('设置服务不可用,无法增强提示词', 'settings-unavailable')
|
|
205
|
+
return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode)
|
|
206
|
+
}
|
|
207
|
+
|
|
199
208
|
const channelsView = (): ChannelsView => {
|
|
200
209
|
const value = resolve()
|
|
201
210
|
return { channels: value.channels, defaultChannelId: value.defaultChannelId }
|
|
@@ -209,6 +218,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
209
218
|
resolveChannels: channelsView,
|
|
210
219
|
autoSave: () => resolve().autoSaveToLibrary,
|
|
211
220
|
budget,
|
|
221
|
+
enhance,
|
|
212
222
|
})
|
|
213
223
|
const disposers = routes.map(route => ctx.webServer.register(route))
|
|
214
224
|
return () => { for (const dispose of disposers) dispose() }
|
|
@@ -225,6 +235,7 @@ export function apply(ctx: Context, config?: Config): void {
|
|
|
225
235
|
defaultChannelId: value.defaultChannelId,
|
|
226
236
|
autoSaveToLibrary: value.autoSaveToLibrary,
|
|
227
237
|
budget,
|
|
238
|
+
enhance,
|
|
228
239
|
}
|
|
229
240
|
}), 'dsh-audiogen: agent audio tools')
|
|
230
241
|
})
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 提示词增强:复用 Agent 当前默认模型(设置「模型」里的 provider/model,
|
|
3
|
+
* 即 agent-default-model 命名空间),宿主端发起一次 LLM 调用把用户 prompt
|
|
4
|
+
* 扩展成更适合生成的任务描述。不新增 API key 配置。
|
|
5
|
+
*
|
|
6
|
+
* 面板「✨ 增强提示词」与 generate_audio 工具的 enhance_prompt 都走这里。
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import type { AudioMode } from './protocol.ts'
|
|
10
|
+
import { AudioGenError } from './audio-engine.ts'
|
|
11
|
+
|
|
12
|
+
/** 按生成模式给出增强指令(系统提示)。 */
|
|
13
|
+
function instructionsFor(mode: AudioMode): string {
|
|
14
|
+
const common = [
|
|
15
|
+
'你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,',
|
|
16
|
+
'请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。',
|
|
17
|
+
'只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。',
|
|
18
|
+
'保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。',
|
|
19
|
+
'描述控制在 200-600 字左右。',
|
|
20
|
+
].join('')
|
|
21
|
+
const perMode: Record<AudioMode, string> = {
|
|
22
|
+
tts: '这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。',
|
|
23
|
+
music: '这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。',
|
|
24
|
+
sfx: '这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。',
|
|
25
|
+
voice_design: '这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。',
|
|
26
|
+
}
|
|
27
|
+
return common + perMode[mode]
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export interface PromptEnhanceDeps {
|
|
31
|
+
/** DSH 设置 seam(读 agent-default-model)。 */
|
|
32
|
+
settings: { describe(options?: { redactSecrets?: boolean }): Array<{ ns: unknown; value?: unknown }> }
|
|
33
|
+
/** 宿主 LLM 运行时访问器(延迟读取,调用时才获取)。 */
|
|
34
|
+
llm?: () => unknown
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
|
|
38
|
+
export async function enhancePromptText(deps: PromptEnhanceDeps, prompt: string, mode: AudioMode): Promise<string> {
|
|
39
|
+
const text = prompt.trim()
|
|
40
|
+
if (text === '') throw new AudioGenError('提示词为空,无法增强', 'enhance-empty-prompt')
|
|
41
|
+
const descriptor = (deps.settings.describe({ redactSecrets: true }) ?? []).find(candidate => String(candidate.ns) === 'agent-default-model')
|
|
42
|
+
const value = (descriptor?.value ?? {}) as { provider?: unknown; model?: unknown }
|
|
43
|
+
const provider = typeof value.provider === 'string' && value.provider.trim() !== '' ? value.provider.trim() : ''
|
|
44
|
+
const model = typeof value.model === 'string' && value.model.trim() !== '' ? value.model.trim() : ''
|
|
45
|
+
if (provider === '' || model === '') {
|
|
46
|
+
throw new AudioGenError('未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型', 'no-default-model')
|
|
47
|
+
}
|
|
48
|
+
const runtime = deps.llm?.() as { stream?: (options: unknown) => AsyncIterable<unknown> } | undefined
|
|
49
|
+
if (runtime === undefined || runtime.stream === undefined) {
|
|
50
|
+
throw new AudioGenError('宿主 LLM 服务不可用(ctx.llm 未注册)', 'llm-unavailable')
|
|
51
|
+
}
|
|
52
|
+
const controller = new AbortController()
|
|
53
|
+
const timer = setTimeout(() => controller.abort(new DOMException('The operation timed out.', 'TimeoutError')), 30_000)
|
|
54
|
+
timer.unref?.()
|
|
55
|
+
let output = ''
|
|
56
|
+
try {
|
|
57
|
+
for await (const chunk of runtime.stream({
|
|
58
|
+
provider,
|
|
59
|
+
model,
|
|
60
|
+
messages: [{ role: 'user', content: text }],
|
|
61
|
+
system: instructionsFor(mode),
|
|
62
|
+
temperature: 0.7,
|
|
63
|
+
maxTokens: 1200,
|
|
64
|
+
signal: controller.signal,
|
|
65
|
+
})) {
|
|
66
|
+
const record = chunk as { type?: string; text?: string; block?: { type?: string; text?: string } }
|
|
67
|
+
if (record.type === 'text-delta' && typeof record.text === 'string') {
|
|
68
|
+
output += record.text
|
|
69
|
+
} else if (record.type === 'block-end' && record.block !== undefined
|
|
70
|
+
&& record.block.type === 'text' && typeof record.block.text === 'string') {
|
|
71
|
+
output += record.block.text
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
} finally {
|
|
75
|
+
clearTimeout(timer)
|
|
76
|
+
}
|
|
77
|
+
const result = stripFences(output.trim())
|
|
78
|
+
if (result === '') {
|
|
79
|
+
throw new AudioGenError('模型未返回增强内容:请检查「设置 → 模型」的默认模型是否可用(或稍后重试)', 'enhance-empty-result')
|
|
80
|
+
}
|
|
81
|
+
return result
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** 去掉模型可能包裹的 ``` 代码围栏。 */
|
|
85
|
+
function stripFences(value: string): string {
|
|
86
|
+
if (value === '') return value
|
|
87
|
+
const withoutFence = value.replace(/^```[a-zA-Z]*\s*\n?/, '').replace(/\n?```\s*$/, '')
|
|
88
|
+
return withoutFence.trim()
|
|
89
|
+
}
|
package/src/protocol.ts
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
|
|
9
9
|
|
|
10
10
|
/** Published package version shared by the host updater and the client UI. */
|
|
11
|
-
export const PLUGIN_VERSION = '0.4.
|
|
11
|
+
export const PLUGIN_VERSION = '0.4.6'
|
|
12
12
|
|
|
13
13
|
/** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
|
|
14
14
|
export const SETTINGS_API = {
|
|
@@ -24,6 +24,9 @@ export const TASK_API = {
|
|
|
24
24
|
cancel: '/api/dsh-audiogen/task/cancel',
|
|
25
25
|
} as const
|
|
26
26
|
|
|
27
|
+
/** Loopback-only prompt enhancement route (uses the agent's default model). */
|
|
28
|
+
export const ENHANCE_API = '/api/dsh-audiogen/prompt/enhance' as const
|
|
29
|
+
|
|
27
30
|
/** Host-mediated built-in provider catalog (channels the user can instantiate). */
|
|
28
31
|
export const PRESETS_API = '/api/dsh-audiogen/presets' as const
|
|
29
32
|
|