dsh-audiogen 0.4.4 → 0.4.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -24,6 +24,8 @@ const SETTINGS_API = {
24
24
  const GENERATE_API = "/api/dsh-audiogen/generate";
25
25
  /** Loopback-only task cancellation route (aborts the host-side upstream call). */
26
26
  const TASK_API = { cancel: "/api/dsh-audiogen/task/cancel" };
27
+ /** Loopback-only prompt enhancement route (uses the agent's default model). */
28
+ const ENHANCE_API = "/api/dsh-audiogen/prompt/enhance";
27
29
  /** Host-mediated built-in provider catalog (channels the user can instantiate). */
28
30
  const PRESETS_API = "/api/dsh-audiogen/presets";
29
31
  /** Host-mediated model/voice discovery endpoint. */
@@ -864,6 +866,66 @@ async function generateAudio(channel, request, signal) {
864
866
  return genericAudio(channel, request, signal);
865
867
  }
866
868
  //#endregion
869
+ //#region src/prompt-enhance.ts
870
+ /** 按生成模式给出增强指令(系统提示)。 */
871
+ function instructionsFor(mode) {
872
+ return [
873
+ "你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,",
874
+ "请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。",
875
+ "只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。",
876
+ "保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。",
877
+ "描述控制在 200-600 字左右。"
878
+ ].join("") + {
879
+ tts: "这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。",
880
+ music: "这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。",
881
+ sfx: "这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。",
882
+ voice_design: "这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。"
883
+ }[mode];
884
+ }
885
+ /** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
886
+ async function enhancePromptText(deps, prompt, mode) {
887
+ const text = prompt.trim();
888
+ if (text === "") throw new AudioGenError("提示词为空,无法增强", "enhance-empty-prompt");
889
+ const value = (deps.settings.describe({ redactSecrets: true }) ?? []).find((candidate) => String(candidate.ns) === "agent-default-model")?.value ?? {};
890
+ const provider = typeof value.provider === "string" && value.provider.trim() !== "" ? value.provider.trim() : "";
891
+ const model = typeof value.model === "string" && value.model.trim() !== "" ? value.model.trim() : "";
892
+ if (provider === "" || model === "") throw new AudioGenError("未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型", "no-default-model");
893
+ const runtime = deps.llm?.();
894
+ if (runtime === void 0 || runtime.stream === void 0) throw new AudioGenError("宿主 LLM 服务不可用(ctx.llm 未注册)", "llm-unavailable");
895
+ const controller = new AbortController();
896
+ const timer = setTimeout(() => controller.abort(new DOMException("The operation timed out.", "TimeoutError")), 3e4);
897
+ timer.unref?.();
898
+ let output = "";
899
+ try {
900
+ for await (const chunk of runtime.stream({
901
+ provider,
902
+ model,
903
+ messages: [{
904
+ role: "user",
905
+ content: text
906
+ }],
907
+ system: instructionsFor(mode),
908
+ temperature: .7,
909
+ maxTokens: 1200,
910
+ signal: controller.signal
911
+ })) {
912
+ const record = chunk;
913
+ if (record.type === "text-delta" && typeof record.text === "string") output += record.text;
914
+ else if (record.type === "block-end" && record.block !== void 0 && record.block.type === "text" && typeof record.block.text === "string") output += record.block.text;
915
+ }
916
+ } finally {
917
+ clearTimeout(timer);
918
+ }
919
+ const result = stripFences(output.trim());
920
+ if (result === "") throw new AudioGenError("模型未返回增强内容:请检查「设置 → 模型」的默认模型是否可用(或稍后重试)", "enhance-empty-result");
921
+ return result;
922
+ }
923
+ /** 去掉模型可能包裹的 ``` 代码围栏。 */
924
+ function stripFences(value) {
925
+ if (value === "") return value;
926
+ return value.replace(/^```[a-zA-Z]*\s*\n?/, "").replace(/\n?```\s*$/, "").trim();
927
+ }
928
+ //#endregion
867
929
  //#region src/audio-presets.ts
868
930
  const AUDIO_PRESETS = [
869
931
  {
@@ -2046,6 +2108,36 @@ function makeRoutes(deps) {
2046
2108
  });
2047
2109
  }
2048
2110
  },
2111
+ {
2112
+ kind: "exact",
2113
+ path: ENHANCE_API,
2114
+ handler: async (req, res) => {
2115
+ if (!guard(req, res, "POST")) return;
2116
+ const body = await readJsonBody(req);
2117
+ const prompt = typeof body?.prompt === "string" ? body.prompt.trim() : "";
2118
+ if (prompt === "") {
2119
+ writeJson(res, 200, {
2120
+ ok: false,
2121
+ code: "bad-request",
2122
+ message: "prompt is required"
2123
+ });
2124
+ return;
2125
+ }
2126
+ const mode = body?.mode === "music" ? "music" : body?.mode === "sfx" ? "sfx" : body?.mode === "voice_design" ? "voice_design" : "tts";
2127
+ try {
2128
+ writeJson(res, 200, {
2129
+ ok: true,
2130
+ enhanced: await deps.enhance(prompt, mode)
2131
+ });
2132
+ } catch (error) {
2133
+ writeJson(res, 200, {
2134
+ ok: false,
2135
+ code: "enhance-failed",
2136
+ message: messageOf(error)
2137
+ });
2138
+ }
2139
+ }
2140
+ },
2049
2141
  {
2050
2142
  kind: "prefix",
2051
2143
  path: AUDIO_API.file,
@@ -2483,6 +2575,10 @@ function registerAgentAudioTools(ctx, resolve) {
2483
2575
  type: "string",
2484
2576
  description: "Optional preview text for voice_design."
2485
2577
  },
2578
+ enhance_prompt: {
2579
+ type: "boolean",
2580
+ description: "Enhance the prompt with the agent default model before generating (uses the configured model settings, no extra key). Best-effort: on failure the original prompt is used."
2581
+ },
2486
2582
  speed: {
2487
2583
  type: "number",
2488
2584
  description: "Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1)."
@@ -2725,6 +2821,9 @@ function registerAgentAudioTools(ctx, resolve) {
2725
2821
  /** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
2726
2822
  const runOne = async (picked) => {
2727
2823
  const request = buildRequest(picked);
2824
+ if (args.enhance_prompt === true && config.enhance !== void 0) try {
2825
+ request.prompt = await config.enhance(request.prompt, request.mode);
2826
+ } catch {}
2728
2827
  try {
2729
2828
  const release = await (config.budget?.acquire(exec.signal) ?? Promise.resolve(() => {}));
2730
2829
  let outputs;
@@ -3137,6 +3236,14 @@ function apply(ctx, config) {
3137
3236
  };
3138
3237
  };
3139
3238
  const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations);
3239
+ const enhance = async (prompt, mode) => {
3240
+ const seam = ctx.get("settings");
3241
+ if (seam?.describe === void 0) throw new AudioGenError("设置服务不可用,无法增强提示词", "settings-unavailable");
3242
+ return enhancePromptText({
3243
+ settings: seam,
3244
+ llm: () => ctx.get("llm")
3245
+ }, prompt, mode);
3246
+ };
3140
3247
  const channelsView = () => {
3141
3248
  const value = resolve();
3142
3249
  return {
@@ -3151,7 +3258,8 @@ function apply(ctx, config) {
3151
3258
  settings: seam,
3152
3259
  resolveChannels: channelsView,
3153
3260
  autoSave: () => resolve().autoSaveToLibrary,
3154
- budget
3261
+ budget,
3262
+ enhance
3155
3263
  }).map((route) => ctx.webServer.register(route));
3156
3264
  return () => {
3157
3265
  for (const dispose of disposers) dispose();
@@ -3167,7 +3275,8 @@ function apply(ctx, config) {
3167
3275
  channels: value.channels,
3168
3276
  defaultChannelId: value.defaultChannelId,
3169
3277
  autoSaveToLibrary: value.autoSaveToLibrary,
3170
- budget
3278
+ budget,
3279
+ enhance
3171
3280
  };
3172
3281
  }), "dsh-audiogen: agent audio tools");
3173
3282
  });
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "dsh-audiogen",
3
3
  "description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
4
- "version": "0.4.4",
4
+ "version": "0.4.6",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",
7
7
  "exports": {
@@ -21,6 +21,8 @@ export interface AgentAudioToolConfig {
21
21
  autoSaveToLibrary: boolean
22
22
  /** 全局并发闸门(与面板路由共享「最大并发生成数」)。 */
23
23
  budget?: GenerationBudget
24
+ /** 提示词增强(复用 Agent 默认模型);enhance_prompt=true 时在生成前调用。 */
25
+ enhance?: (prompt: string, mode: AudioMode) => Promise<string>
24
26
  }
25
27
 
26
28
  interface AgentAudioRef {
@@ -157,6 +159,7 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
157
159
  },
158
160
  voice: { type: 'string', description: 'Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio.' },
159
161
  preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
162
+ enhance_prompt: { type: 'boolean', description: 'Enhance the prompt with the agent default model before generating (uses the configured model settings, no extra key). Best-effort: on failure the original prompt is used.' },
160
163
  speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1).' },
161
164
  duration: { type: 'number', description: 'Requested duration in seconds for music/sfx.' },
162
165
  lyrics: { type: 'string', description: 'Lyrics for music generation (MiniMax music-3.0/music-cover). Required unless is_instrumental is true. Split verses with an empty line.' },
@@ -302,6 +305,14 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
302
305
  /** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
303
306
  const runOne = async (picked: { channel: AudioChannel; alias: string; upstream: string }): Promise<AgentAudioGroup> => {
304
307
  const request = buildRequest(picked)
308
+ // 可选:生成前用 Agent 默认模型增强 prompt(失败则沿用原文)
309
+ if (args.enhance_prompt === true && config.enhance !== undefined) {
310
+ try {
311
+ request.prompt = await config.enhance(request.prompt, request.mode)
312
+ } catch {
313
+ // 增强失败不阻断生成
314
+ }
315
+ }
305
316
  try {
306
317
  // 与面板路由共享全局并发闸门(限流时排队;取消时立即出队)。
307
318
  const release = await (config.budget?.acquire(exec.signal) ?? Promise.resolve(() => { /* 默认不限制 */ }))
package/src/client/api.ts CHANGED
@@ -4,7 +4,7 @@
4
4
  */
5
5
 
6
6
  import {
7
- GENERATE_API, HISTORY_API, LIBRARY_API, TASK_API,
7
+ ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, TASK_API,
8
8
  type GenerateAudioRequest, type GeneratedAudio, type HistoryEntry,
9
9
  type LibraryEntry, type LibrarySaveRequest, type LibraryUpdateRequest,
10
10
  } from '../protocol.ts'
@@ -42,6 +42,13 @@ export class AudiogenApi {
42
42
  await postJson(TASK_API.cancel, { taskId }).catch(() => { /* best-effort */ })
43
43
  }
44
44
 
45
+ /** 提示词增强(复用 Agent 默认模型)。 */
46
+ async enhancePrompt(prompt: string, mode: string): Promise<{ ok: boolean; enhanced?: string; code?: string; message?: string }> {
47
+ const response = await postJson(ENHANCE_API, { prompt, mode })
48
+ const body = await response.json() as { ok?: boolean; enhanced?: string; code?: string; message?: string }
49
+ return { ok: body.ok === true, ...(body.enhanced === undefined ? {} : { enhanced: body.enhanced }), ...(body.code === undefined ? {} : { code: body.code }), ...(body.message === undefined ? {} : { message: body.message }) }
50
+ }
51
+
45
52
  async history(): Promise<HistoryEntry[]> {
46
53
  const response = await postJson(HISTORY_API.list, {})
47
54
  const body = await response.json() as { ok?: boolean; history?: HistoryEntry[] }
@@ -97,9 +97,9 @@
97
97
  flex-direction: column;
98
98
  gap: 11px;
99
99
  flex: none;
100
- width: 300px;
101
- min-width: 260px;
102
- max-width: 340px;
100
+ width: 380px;
101
+ min-width: 320px;
102
+ max-width: 440px;
103
103
  min-height: 0;
104
104
  overflow-y: auto;
105
105
  padding: 2px;
@@ -111,8 +111,8 @@
111
111
  .formCol::-webkit-scrollbar-thumb { background: var(--dsw-alias-border-l2); border-radius: 999px; }
112
112
 
113
113
  .modeRow {
114
- display: grid;
115
- grid-template-columns: repeat(4, minmax(0, 1fr));
114
+ display: flex;
115
+ flex-wrap: wrap;
116
116
  gap: 6px;
117
117
  }
118
118
 
@@ -120,8 +120,10 @@
120
120
  display: flex;
121
121
  align-items: center;
122
122
  justify-content: center;
123
+ flex: 1 1 auto;
124
+ min-width: 0;
123
125
  min-height: 34px;
124
- padding: 6px 4px;
126
+ padding: 6px 10px;
125
127
  border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
126
128
  border-radius: 9px;
127
129
  background: var(--dsw-alias-bg-layer-1, #fff);
@@ -258,6 +260,7 @@
258
260
  .resultCol {
259
261
  flex: 1;
260
262
  min-width: 0;
263
+ max-width: 760px;
261
264
  display: flex;
262
265
  flex-direction: column;
263
266
  gap: 10px;
@@ -960,7 +963,7 @@
960
963
 
961
964
  .resultGroupError {
962
965
  font-size: 12px;
963
- color: #dc2626;
966
+ color: var(--dsw-alias-state-error-primary, #dc2626);
964
967
  }
965
968
 
966
969
  .resultGroupCount {
@@ -1021,7 +1024,7 @@
1021
1024
  }
1022
1025
 
1023
1026
  .taskCard[data-state='failed'] {
1024
- border-color: #dc2626;
1027
+ border-color: var(--dsw-alias-state-error-primary, #dc2626);
1025
1028
  }
1026
1029
 
1027
1030
  .taskCard[data-state='cancelled'] {
@@ -1051,11 +1054,11 @@
1051
1054
  }
1052
1055
 
1053
1056
  .taskStatus[data-state='done'] {
1054
- color: #16a34a;
1057
+ color: var(--dsw-alias-state-success-primary, #16a34a);
1055
1058
  }
1056
1059
 
1057
1060
  .taskStatus[data-state='failed'] {
1058
- color: #dc2626;
1061
+ color: var(--dsw-alias-state-error-primary, #dc2626);
1059
1062
  }
1060
1063
 
1061
1064
  .taskActions {
@@ -1107,11 +1110,11 @@
1107
1110
  .historyCompareBadge {
1108
1111
  font-size: 11px;
1109
1112
  font-weight: 600;
1110
- color: #7c3aed;
1111
- border: 1px solid #ddd6fe;
1113
+ color: var(--dsw-alias-label-secondary, #6b7280);
1114
+ border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
1112
1115
  border-radius: 999px;
1113
1116
  padding: 1px 8px;
1114
- background: #f5f3ff;
1117
+ background: var(--dsw-alias-bg-layer-3, #fff);
1115
1118
  white-space: nowrap;
1116
1119
  }
1117
1120
 
@@ -1221,7 +1224,7 @@
1221
1224
  }
1222
1225
 
1223
1226
  .historyIcon[title^='删除']:hover {
1224
- color: #dc2626;
1227
+ color: var(--dsw-alias-state-error-primary, #dc2626);
1225
1228
  }
1226
1229
 
1227
1230
  /* 历史 prompt 两行截断 */
@@ -1231,3 +1234,90 @@
1231
1234
  -webkit-box-orient: vertical;
1232
1235
  overflow: hidden;
1233
1236
  }
1237
+
1238
+ /* ------------------------------------------------------- P3 打磨:响应式与统一质感 */
1239
+
1240
+ /* 焦点态统一(键盘可达性) */
1241
+ .input:focus-visible,
1242
+ .select:focus-visible,
1243
+ .textarea:focus-visible,
1244
+ .modeButton:focus-visible,
1245
+ .ghostButton:focus-visible,
1246
+ .historyAction:focus-visible,
1247
+ .historyIcon:focus-visible {
1248
+ outline: 2px solid var(--dsw-alias-interactive-bg-hover-accent, rgba(38, 49, 72, 0.35));
1249
+ outline-offset: 1px;
1250
+ }
1251
+
1252
+ .generate:focus-visible {
1253
+ outline: 2px solid var(--dsw-alias-interactive-bg-hover-accent, rgba(38, 49, 72, 0.35));
1254
+ outline-offset: 2px;
1255
+ }
1256
+
1257
+ /* 中窄屏:历史移到下方整行 */
1258
+ @media (max-width: 1100px) {
1259
+ .studio {
1260
+ flex-wrap: wrap;
1261
+ }
1262
+
1263
+ .historyCol {
1264
+ width: 100%;
1265
+ min-width: 0;
1266
+ max-width: none;
1267
+ }
1268
+ }
1269
+
1270
+ /* 窄屏:单列堆叠 */
1271
+ @media (max-width: 720px) {
1272
+ .formCol {
1273
+ width: 100%;
1274
+ min-width: 0;
1275
+ max-width: none;
1276
+ }
1277
+
1278
+ .resultCol {
1279
+ min-height: 320px;
1280
+ }
1281
+ }
1282
+
1283
+ /* 提示词增强 */
1284
+ .formSectionRow {
1285
+ display: flex;
1286
+ align-items: flex-end;
1287
+ justify-content: space-between;
1288
+ gap: 8px;
1289
+ }
1290
+
1291
+ .formSectionRow .formSection {
1292
+ flex: 1;
1293
+ }
1294
+
1295
+ .enhanceCard {
1296
+ border: 1px solid var(--dsw-alias-border-l1, #e5e7eb);
1297
+ border-radius: 10px;
1298
+ padding: 10px;
1299
+ background: var(--dsw-alias-bg-layer-2, #fafafa);
1300
+ display: flex;
1301
+ flex-direction: column;
1302
+ gap: 8px;
1303
+ }
1304
+
1305
+ .enhanceCardHead {
1306
+ display: flex;
1307
+ align-items: center;
1308
+ justify-content: space-between;
1309
+ gap: 8px;
1310
+ font-size: 12px;
1311
+ color: var(--dsw-alias-label-secondary, #6b7280);
1312
+ }
1313
+
1314
+ .enhanceActions {
1315
+ display: flex;
1316
+ gap: 6px;
1317
+ }
1318
+
1319
+ .enhanceActionsRow {
1320
+ display: flex;
1321
+ justify-content: flex-end;
1322
+ margin-top: -4px;
1323
+ }
@@ -230,6 +230,9 @@ export function StudioView(props: {
230
230
  const [bitrate, setBitrate] = useState('')
231
231
  const [audioChannel, setAudioChannel] = useState('')
232
232
  const [subtitle, setSubtitle] = useState(false)
233
+ // 提示词增强
234
+ const [enhancing, setEnhancing] = useState(false)
235
+ const [enhancePreview, setEnhancePreview] = useState<string | null>(null)
233
236
  // Stable Audio 参数(仅 Stability 渠道显示)
234
237
  const [seed, setSeed] = useState('')
235
238
  const [steps, setSteps] = useState('')
@@ -524,6 +527,8 @@ export function StudioView(props: {
524
527
  }
525
528
  const bool = (key: string): boolean | undefined => typeof params[key] === 'boolean' ? params[key] as boolean : undefined
526
529
  setMode(modeValue)
530
+ const promptValue = str('prompt')
531
+ if (promptValue !== undefined) setPrompt(promptValue)
527
532
  const modelValue = str('model') ?? singleModel
528
533
  if (modelValue !== '') setModel(modelValue)
529
534
  if (compareModelsRestore !== undefined && compareModelsRestore.length > 0) {
@@ -578,6 +583,28 @@ export function StudioView(props: {
578
583
  props.showToast('已恢复该次生成的配置,可直接再次生成')
579
584
  }
580
585
 
586
+ /** 调用宿主增强(Agent 默认模型),结果先预览再应用。 */
587
+ const runEnhance = async (): Promise<void> => {
588
+ if (prompt.trim() === '') {
589
+ setError('请先输入文本/提示词,再点击增强')
590
+ return
591
+ }
592
+ setEnhancing(true)
593
+ setError(null)
594
+ try {
595
+ const result = await api.enhancePrompt(prompt.trim(), mode)
596
+ if (result.ok !== true || result.enhanced === undefined || result.enhanced.trim() === '') {
597
+ setError(result.message ?? '增强失败,请稍后重试')
598
+ return
599
+ }
600
+ setEnhancePreview(result.enhanced.trim())
601
+ } catch (err) {
602
+ setError(err instanceof Error ? err.message : String(err))
603
+ } finally {
604
+ setEnhancing(false)
605
+ }
606
+ }
607
+
581
608
  /** 删除历史记录(对比任务卡删除该任务的全部模型条目)。 */
582
609
  const deleteHistoryEntries = async (ids: string[]): Promise<void> => {
583
610
  try {
@@ -875,6 +902,24 @@ export function StudioView(props: {
875
902
  <span>{mode === 'voice_design' ? '音色描述' : mode === 'tts' ? '文本' : '提示词'}</span>
876
903
  <textarea className={css.textarea} value={prompt} onChange={event => setPrompt(event.target.value)} placeholder={tt('prompt.placeholder')} />
877
904
  </label>
905
+ <div className={css.enhanceActionsRow}>
906
+ <button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>
907
+ {enhancing ? '增强中…' : '✨ 增强提示词'}
908
+ </button>
909
+ </div>
910
+ {enhancePreview !== null ? (
911
+ <div className={css.enhanceCard}>
912
+ <div className={css.enhanceCardHead}>
913
+ <strong>增强结果({modeLabelOf(mode)})</strong>
914
+ <span className={css.enhanceActions}>
915
+ <button type="button" className={css.ghostButton} onClick={() => { setPrompt(enhancePreview); setEnhancePreview(null); props.showToast('已应用增强结果') }}>应用</button>
916
+ <button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>重新生成</button>
917
+ <button type="button" className={css.ghostButton} onClick={() => setEnhancePreview(null)}>放弃</button>
918
+ </span>
919
+ </div>
920
+ <textarea className={css.textarea} value={enhancePreview} readOnly />
921
+ </div>
922
+ ) : null}
878
923
 
879
924
  {mode === 'voice_design' ? (
880
925
  <>
package/src/index.ts CHANGED
@@ -16,8 +16,10 @@ import z from 'schemastery'
16
16
  import type {} from '@deepseek-ai/dsh-host-webserver'
17
17
  import type {} from '@deepseek-ai/dsh-system-prompt'
18
18
  import type {} from '@deepseek-ai/dsh-tools'
19
- import { AUDIOGEN_SETTINGS_NAMESPACE, type ChannelConfig, type ModelMapping } from './protocol.ts'
19
+ import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type ModelMapping } from './protocol.ts'
20
20
  import { createGenerationBudget } from './audio-scheduler.ts'
21
+ import { enhancePromptText } from './prompt-enhance.ts'
22
+ import { AudioGenError } from './audio-engine.ts'
21
23
  import { makeRoutes, type ChannelsView, type SettingsSeam } from './routes.ts'
22
24
  import type { AudioChannel } from './audio-engine.ts'
23
25
  import { registerAgentAudioTools, type AgentAudioToolConfig } from './agent-audio-tools.ts'
@@ -196,6 +198,13 @@ export function apply(ctx: Context, config?: Config): void {
196
198
  // 全局并发闸门:所有上游调用(面板路由 + Agent 工具)共享「最大并发生成数」。
197
199
  const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations)
198
200
 
201
+ // 提示词增强:复用 Agent 默认模型(agent-default-model 设置),面板与 Agent 工具共用。
202
+ const enhance = async (prompt: string, mode: AudioMode): Promise<string> => {
203
+ const seam = ctx.get('settings') as unknown as SettingsSeam
204
+ if (seam?.describe === undefined) throw new AudioGenError('设置服务不可用,无法增强提示词', 'settings-unavailable')
205
+ return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode)
206
+ }
207
+
199
208
  const channelsView = (): ChannelsView => {
200
209
  const value = resolve()
201
210
  return { channels: value.channels, defaultChannelId: value.defaultChannelId }
@@ -209,6 +218,7 @@ export function apply(ctx: Context, config?: Config): void {
209
218
  resolveChannels: channelsView,
210
219
  autoSave: () => resolve().autoSaveToLibrary,
211
220
  budget,
221
+ enhance,
212
222
  })
213
223
  const disposers = routes.map(route => ctx.webServer.register(route))
214
224
  return () => { for (const dispose of disposers) dispose() }
@@ -225,6 +235,7 @@ export function apply(ctx: Context, config?: Config): void {
225
235
  defaultChannelId: value.defaultChannelId,
226
236
  autoSaveToLibrary: value.autoSaveToLibrary,
227
237
  budget,
238
+ enhance,
228
239
  }
229
240
  }), 'dsh-audiogen: agent audio tools')
230
241
  })
@@ -0,0 +1,89 @@
1
+ /**
2
+ * 提示词增强:复用 Agent 当前默认模型(设置「模型」里的 provider/model,
3
+ * 即 agent-default-model 命名空间),宿主端发起一次 LLM 调用把用户 prompt
4
+ * 扩展成更适合生成的任务描述。不新增 API key 配置。
5
+ *
6
+ * 面板「✨ 增强提示词」与 generate_audio 工具的 enhance_prompt 都走这里。
7
+ */
8
+
9
+ import type { AudioMode } from './protocol.ts'
10
+ import { AudioGenError } from './audio-engine.ts'
11
+
12
+ /** 按生成模式给出增强指令(系统提示)。 */
13
+ function instructionsFor(mode: AudioMode): string {
14
+ const common = [
15
+ '你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,',
16
+ '请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。',
17
+ '只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。',
18
+ '保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。',
19
+ '描述控制在 200-600 字左右。',
20
+ ].join('')
21
+ const perMode: Record<AudioMode, string> = {
22
+ tts: '这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。',
23
+ music: '这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。',
24
+ sfx: '这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。',
25
+ voice_design: '这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。',
26
+ }
27
+ return common + perMode[mode]
28
+ }
29
+
30
+ export interface PromptEnhanceDeps {
31
+ /** DSH 设置 seam(读 agent-default-model)。 */
32
+ settings: { describe(options?: { redactSecrets?: boolean }): Array<{ ns: unknown; value?: unknown }> }
33
+ /** 宿主 LLM 运行时访问器(延迟读取,调用时才获取)。 */
34
+ llm?: () => unknown
35
+ }
36
+
37
+ /** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
38
+ export async function enhancePromptText(deps: PromptEnhanceDeps, prompt: string, mode: AudioMode): Promise<string> {
39
+ const text = prompt.trim()
40
+ if (text === '') throw new AudioGenError('提示词为空,无法增强', 'enhance-empty-prompt')
41
+ const descriptor = (deps.settings.describe({ redactSecrets: true }) ?? []).find(candidate => String(candidate.ns) === 'agent-default-model')
42
+ const value = (descriptor?.value ?? {}) as { provider?: unknown; model?: unknown }
43
+ const provider = typeof value.provider === 'string' && value.provider.trim() !== '' ? value.provider.trim() : ''
44
+ const model = typeof value.model === 'string' && value.model.trim() !== '' ? value.model.trim() : ''
45
+ if (provider === '' || model === '') {
46
+ throw new AudioGenError('未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型', 'no-default-model')
47
+ }
48
+ const runtime = deps.llm?.() as { stream?: (options: unknown) => AsyncIterable<unknown> } | undefined
49
+ if (runtime === undefined || runtime.stream === undefined) {
50
+ throw new AudioGenError('宿主 LLM 服务不可用(ctx.llm 未注册)', 'llm-unavailable')
51
+ }
52
+ const controller = new AbortController()
53
+ const timer = setTimeout(() => controller.abort(new DOMException('The operation timed out.', 'TimeoutError')), 30_000)
54
+ timer.unref?.()
55
+ let output = ''
56
+ try {
57
+ for await (const chunk of runtime.stream({
58
+ provider,
59
+ model,
60
+ messages: [{ role: 'user', content: text }],
61
+ system: instructionsFor(mode),
62
+ temperature: 0.7,
63
+ maxTokens: 1200,
64
+ signal: controller.signal,
65
+ })) {
66
+ const record = chunk as { type?: string; text?: string; block?: { type?: string; text?: string } }
67
+ if (record.type === 'text-delta' && typeof record.text === 'string') {
68
+ output += record.text
69
+ } else if (record.type === 'block-end' && record.block !== undefined
70
+ && record.block.type === 'text' && typeof record.block.text === 'string') {
71
+ output += record.block.text
72
+ }
73
+ }
74
+ } finally {
75
+ clearTimeout(timer)
76
+ }
77
+ const result = stripFences(output.trim())
78
+ if (result === '') {
79
+ throw new AudioGenError('模型未返回增强内容:请检查「设置 → 模型」的默认模型是否可用(或稍后重试)', 'enhance-empty-result')
80
+ }
81
+ return result
82
+ }
83
+
84
+ /** 去掉模型可能包裹的 ``` 代码围栏。 */
85
+ function stripFences(value: string): string {
86
+ if (value === '') return value
87
+ const withoutFence = value.replace(/^```[a-zA-Z]*\s*\n?/, '').replace(/\n?```\s*$/, '')
88
+ return withoutFence.trim()
89
+ }
package/src/protocol.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
9
9
 
10
10
  /** Published package version shared by the host updater and the client UI. */
11
- export const PLUGIN_VERSION = '0.4.4'
11
+ export const PLUGIN_VERSION = '0.4.6'
12
12
 
13
13
  /** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
14
14
  export const SETTINGS_API = {
@@ -24,6 +24,9 @@ export const TASK_API = {
24
24
  cancel: '/api/dsh-audiogen/task/cancel',
25
25
  } as const
26
26
 
27
+ /** Loopback-only prompt enhancement route (uses the agent's default model). */
28
+ export const ENHANCE_API = '/api/dsh-audiogen/prompt/enhance' as const
29
+
27
30
  /** Host-mediated built-in provider catalog (channels the user can instantiate). */
28
31
  export const PRESETS_API = '/api/dsh-audiogen/presets' as const
29
32