dsh-audiogen 0.4.4 → 0.4.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -24,6 +24,8 @@ const SETTINGS_API = {
24
24
  const GENERATE_API = "/api/dsh-audiogen/generate";
25
25
  /** Loopback-only task cancellation route (aborts the host-side upstream call). */
26
26
  const TASK_API = { cancel: "/api/dsh-audiogen/task/cancel" };
27
+ /** Loopback-only prompt enhancement route (uses the agent's default model). */
28
+ const ENHANCE_API = "/api/dsh-audiogen/prompt/enhance";
27
29
  /** Host-mediated built-in provider catalog (channels the user can instantiate). */
28
30
  const PRESETS_API = "/api/dsh-audiogen/presets";
29
31
  /** Host-mediated model/voice discovery endpoint. */
@@ -864,6 +866,55 @@ async function generateAudio(channel, request, signal) {
864
866
  return genericAudio(channel, request, signal);
865
867
  }
866
868
  //#endregion
869
+ //#region src/prompt-enhance.ts
870
+ /** 按生成模式给出增强指令(系统提示)。 */
871
+ function instructionsFor(mode) {
872
+ return [
873
+ "你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,",
874
+ "请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。",
875
+ "只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。",
876
+ "保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。",
877
+ "描述控制在 200-600 字左右。"
878
+ ].join("") + {
879
+ tts: "这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。",
880
+ music: "这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。",
881
+ sfx: "这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。",
882
+ voice_design: "这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。"
883
+ }[mode];
884
+ }
885
+ /** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
886
+ async function enhancePromptText(deps, prompt, mode) {
887
+ const text = prompt.trim();
888
+ if (text === "") throw new AudioGenError("提示词为空,无法增强", "enhance-empty-prompt");
889
+ const value = (deps.settings.describe({ redactSecrets: true }) ?? []).find((candidate) => String(candidate.ns) === "agent-default-model")?.value ?? {};
890
+ const provider = typeof value.provider === "string" && value.provider.trim() !== "" ? value.provider.trim() : "";
891
+ const model = typeof value.model === "string" && value.model.trim() !== "" ? value.model.trim() : "";
892
+ if (provider === "" || model === "") throw new AudioGenError("未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型", "no-default-model");
893
+ const runtime = deps.llm?.();
894
+ if (runtime === void 0 || runtime.stream === void 0) throw new AudioGenError("宿主 LLM 服务不可用(ctx.llm 未注册)", "llm-unavailable");
895
+ let output = "";
896
+ for await (const chunk of runtime.stream({
897
+ provider,
898
+ model,
899
+ messages: [{
900
+ role: "user",
901
+ content: text
902
+ }],
903
+ system: instructionsFor(mode),
904
+ temperature: .7,
905
+ maxTokens: 1200
906
+ })) {
907
+ const record = chunk;
908
+ if (record.type === "text-delta" && typeof record.text === "string") output += record.text;
909
+ }
910
+ return stripFences(output.trim());
911
+ }
912
+ /** 去掉模型可能包裹的 ``` 代码围栏。 */
913
+ function stripFences(value) {
914
+ if (value === "") return value;
915
+ return value.replace(/^```[a-zA-Z]*\s*\n?/, "").replace(/\n?```\s*$/, "").trim();
916
+ }
917
+ //#endregion
867
918
  //#region src/audio-presets.ts
868
919
  const AUDIO_PRESETS = [
869
920
  {
@@ -2046,6 +2097,36 @@ function makeRoutes(deps) {
2046
2097
  });
2047
2098
  }
2048
2099
  },
2100
+ {
2101
+ kind: "exact",
2102
+ path: ENHANCE_API,
2103
+ handler: async (req, res) => {
2104
+ if (!guard(req, res, "POST")) return;
2105
+ const body = await readJsonBody(req);
2106
+ const prompt = typeof body?.prompt === "string" ? body.prompt.trim() : "";
2107
+ if (prompt === "") {
2108
+ writeJson(res, 200, {
2109
+ ok: false,
2110
+ code: "bad-request",
2111
+ message: "prompt is required"
2112
+ });
2113
+ return;
2114
+ }
2115
+ const mode = body?.mode === "music" ? "music" : body?.mode === "sfx" ? "sfx" : body?.mode === "voice_design" ? "voice_design" : "tts";
2116
+ try {
2117
+ writeJson(res, 200, {
2118
+ ok: true,
2119
+ enhanced: await deps.enhance(prompt, mode)
2120
+ });
2121
+ } catch (error) {
2122
+ writeJson(res, 200, {
2123
+ ok: false,
2124
+ code: "enhance-failed",
2125
+ message: messageOf(error)
2126
+ });
2127
+ }
2128
+ }
2129
+ },
2049
2130
  {
2050
2131
  kind: "prefix",
2051
2132
  path: AUDIO_API.file,
@@ -2483,6 +2564,10 @@ function registerAgentAudioTools(ctx, resolve) {
2483
2564
  type: "string",
2484
2565
  description: "Optional preview text for voice_design."
2485
2566
  },
2567
+ enhance_prompt: {
2568
+ type: "boolean",
2569
+ description: "Enhance the prompt with the agent default model before generating (uses the configured model settings, no extra key). Best-effort: on failure the original prompt is used."
2570
+ },
2486
2571
  speed: {
2487
2572
  type: "number",
2488
2573
  description: "Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1)."
@@ -2725,6 +2810,9 @@ function registerAgentAudioTools(ctx, resolve) {
2725
2810
  /** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
2726
2811
  const runOne = async (picked) => {
2727
2812
  const request = buildRequest(picked);
2813
+ if (args.enhance_prompt === true && config.enhance !== void 0) try {
2814
+ request.prompt = await config.enhance(request.prompt, request.mode);
2815
+ } catch {}
2728
2816
  try {
2729
2817
  const release = await (config.budget?.acquire(exec.signal) ?? Promise.resolve(() => {}));
2730
2818
  let outputs;
@@ -3137,6 +3225,14 @@ function apply(ctx, config) {
3137
3225
  };
3138
3226
  };
3139
3227
  const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations);
3228
+ const enhance = async (prompt, mode) => {
3229
+ const seam = ctx.get("settings");
3230
+ if (seam?.describe === void 0) throw new AudioGenError("设置服务不可用,无法增强提示词", "settings-unavailable");
3231
+ return enhancePromptText({
3232
+ settings: seam,
3233
+ llm: () => ctx.get("llm")
3234
+ }, prompt, mode);
3235
+ };
3140
3236
  const channelsView = () => {
3141
3237
  const value = resolve();
3142
3238
  return {
@@ -3151,7 +3247,8 @@ function apply(ctx, config) {
3151
3247
  settings: seam,
3152
3248
  resolveChannels: channelsView,
3153
3249
  autoSave: () => resolve().autoSaveToLibrary,
3154
- budget
3250
+ budget,
3251
+ enhance
3155
3252
  }).map((route) => ctx.webServer.register(route));
3156
3253
  return () => {
3157
3254
  for (const dispose of disposers) dispose();
@@ -3167,7 +3264,8 @@ function apply(ctx, config) {
3167
3264
  channels: value.channels,
3168
3265
  defaultChannelId: value.defaultChannelId,
3169
3266
  autoSaveToLibrary: value.autoSaveToLibrary,
3170
- budget
3267
+ budget,
3268
+ enhance
3171
3269
  };
3172
3270
  }), "dsh-audiogen: agent audio tools");
3173
3271
  });
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "dsh-audiogen",
3
3
  "description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
4
- "version": "0.4.4",
4
+ "version": "0.4.5",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",
7
7
  "exports": {
@@ -21,6 +21,8 @@ export interface AgentAudioToolConfig {
21
21
  autoSaveToLibrary: boolean
22
22
  /** 全局并发闸门(与面板路由共享「最大并发生成数」)。 */
23
23
  budget?: GenerationBudget
24
+ /** 提示词增强(复用 Agent 默认模型);enhance_prompt=true 时在生成前调用。 */
25
+ enhance?: (prompt: string, mode: AudioMode) => Promise<string>
24
26
  }
25
27
 
26
28
  interface AgentAudioRef {
@@ -157,6 +159,7 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
157
159
  },
158
160
  voice: { type: 'string', description: 'Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio.' },
159
161
  preview_text: { type: 'string', description: 'Optional preview text for voice_design.' },
162
+ enhance_prompt: { type: 'boolean', description: 'Enhance the prompt with the agent default model before generating (uses the configured model settings, no extra key). Best-effort: on failure the original prompt is used.' },
160
163
  speed: { type: 'number', description: 'Optional speaking rate / speed multiplier where supported. MiniMax range 0.5-2.0 (default 1).' },
161
164
  duration: { type: 'number', description: 'Requested duration in seconds for music/sfx.' },
162
165
  lyrics: { type: 'string', description: 'Lyrics for music generation (MiniMax music-3.0/music-cover). Required unless is_instrumental is true. Split verses with an empty line.' },
@@ -302,6 +305,14 @@ export function registerAgentAudioTools(ctx: Context, resolve: () => AgentAudioT
302
305
  /** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
303
306
  const runOne = async (picked: { channel: AudioChannel; alias: string; upstream: string }): Promise<AgentAudioGroup> => {
304
307
  const request = buildRequest(picked)
308
+ // 可选:生成前用 Agent 默认模型增强 prompt(失败则沿用原文)
309
+ if (args.enhance_prompt === true && config.enhance !== undefined) {
310
+ try {
311
+ request.prompt = await config.enhance(request.prompt, request.mode)
312
+ } catch {
313
+ // 增强失败不阻断生成
314
+ }
315
+ }
305
316
  try {
306
317
  // 与面板路由共享全局并发闸门(限流时排队;取消时立即出队)。
307
318
  const release = await (config.budget?.acquire(exec.signal) ?? Promise.resolve(() => { /* 默认不限制 */ }))
package/src/client/api.ts CHANGED
@@ -4,7 +4,7 @@
4
4
  */
5
5
 
6
6
  import {
7
- GENERATE_API, HISTORY_API, LIBRARY_API, TASK_API,
7
+ ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, TASK_API,
8
8
  type GenerateAudioRequest, type GeneratedAudio, type HistoryEntry,
9
9
  type LibraryEntry, type LibrarySaveRequest, type LibraryUpdateRequest,
10
10
  } from '../protocol.ts'
@@ -42,6 +42,13 @@ export class AudiogenApi {
42
42
  await postJson(TASK_API.cancel, { taskId }).catch(() => { /* best-effort */ })
43
43
  }
44
44
 
45
+ /** 提示词增强(复用 Agent 默认模型)。 */
46
+ async enhancePrompt(prompt: string, mode: string): Promise<{ ok: boolean; enhanced?: string; code?: string; message?: string }> {
47
+ const response = await postJson(ENHANCE_API, { prompt, mode })
48
+ const body = await response.json() as { ok?: boolean; enhanced?: string; code?: string; message?: string }
49
+ return { ok: body.ok === true, ...(body.enhanced === undefined ? {} : { enhanced: body.enhanced }), ...(body.code === undefined ? {} : { code: body.code }), ...(body.message === undefined ? {} : { message: body.message }) }
50
+ }
51
+
45
52
  async history(): Promise<HistoryEntry[]> {
46
53
  const response = await postJson(HISTORY_API.list, {})
47
54
  const body = await response.json() as { ok?: boolean; history?: HistoryEntry[] }
@@ -111,8 +111,8 @@
111
111
  .formCol::-webkit-scrollbar-thumb { background: var(--dsw-alias-border-l2); border-radius: 999px; }
112
112
 
113
113
  .modeRow {
114
- display: grid;
115
- grid-template-columns: repeat(4, minmax(0, 1fr));
114
+ display: flex;
115
+ flex-wrap: wrap;
116
116
  gap: 6px;
117
117
  }
118
118
 
@@ -120,8 +120,10 @@
120
120
  display: flex;
121
121
  align-items: center;
122
122
  justify-content: center;
123
+ flex: 1 1 auto;
124
+ min-width: 0;
123
125
  min-height: 34px;
124
- padding: 6px 4px;
126
+ padding: 6px 10px;
125
127
  border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
126
128
  border-radius: 9px;
127
129
  background: var(--dsw-alias-bg-layer-1, #fff);
@@ -960,7 +962,7 @@
960
962
 
961
963
  .resultGroupError {
962
964
  font-size: 12px;
963
- color: #dc2626;
965
+ color: var(--dsw-alias-state-error-primary, #dc2626);
964
966
  }
965
967
 
966
968
  .resultGroupCount {
@@ -1021,7 +1023,7 @@
1021
1023
  }
1022
1024
 
1023
1025
  .taskCard[data-state='failed'] {
1024
- border-color: #dc2626;
1026
+ border-color: var(--dsw-alias-state-error-primary, #dc2626);
1025
1027
  }
1026
1028
 
1027
1029
  .taskCard[data-state='cancelled'] {
@@ -1051,11 +1053,11 @@
1051
1053
  }
1052
1054
 
1053
1055
  .taskStatus[data-state='done'] {
1054
- color: #16a34a;
1056
+ color: var(--dsw-alias-state-success-primary, #16a34a);
1055
1057
  }
1056
1058
 
1057
1059
  .taskStatus[data-state='failed'] {
1058
- color: #dc2626;
1060
+ color: var(--dsw-alias-state-error-primary, #dc2626);
1059
1061
  }
1060
1062
 
1061
1063
  .taskActions {
@@ -1107,11 +1109,11 @@
1107
1109
  .historyCompareBadge {
1108
1110
  font-size: 11px;
1109
1111
  font-weight: 600;
1110
- color: #7c3aed;
1111
- border: 1px solid #ddd6fe;
1112
+ color: var(--dsw-alias-label-secondary, #6b7280);
1113
+ border: 1px solid var(--dsw-alias-border-l2, #d1d5db);
1112
1114
  border-radius: 999px;
1113
1115
  padding: 1px 8px;
1114
- background: #f5f3ff;
1116
+ background: var(--dsw-alias-bg-layer-3, #fff);
1115
1117
  white-space: nowrap;
1116
1118
  }
1117
1119
 
@@ -1221,7 +1223,7 @@
1221
1223
  }
1222
1224
 
1223
1225
  .historyIcon[title^='删除']:hover {
1224
- color: #dc2626;
1226
+ color: var(--dsw-alias-state-error-primary, #dc2626);
1225
1227
  }
1226
1228
 
1227
1229
  /* 历史 prompt 两行截断 */
@@ -1231,3 +1233,84 @@
1231
1233
  -webkit-box-orient: vertical;
1232
1234
  overflow: hidden;
1233
1235
  }
1236
+
1237
+ /* ------------------------------------------------------- P3 打磨:响应式与统一质感 */
1238
+
1239
+ /* 焦点态统一(键盘可达性) */
1240
+ .input:focus-visible,
1241
+ .select:focus-visible,
1242
+ .textarea:focus-visible,
1243
+ .modeButton:focus-visible,
1244
+ .ghostButton:focus-visible,
1245
+ .historyAction:focus-visible,
1246
+ .historyIcon:focus-visible {
1247
+ outline: 2px solid var(--dsw-alias-interactive-bg-hover-accent, rgba(38, 49, 72, 0.35));
1248
+ outline-offset: 1px;
1249
+ }
1250
+
1251
+ .generate:focus-visible {
1252
+ outline: 2px solid var(--dsw-alias-interactive-bg-hover-accent, rgba(38, 49, 72, 0.35));
1253
+ outline-offset: 2px;
1254
+ }
1255
+
1256
+ /* 中窄屏:历史移到下方整行 */
1257
+ @media (max-width: 1100px) {
1258
+ .studio {
1259
+ flex-wrap: wrap;
1260
+ }
1261
+
1262
+ .historyCol {
1263
+ width: 100%;
1264
+ min-width: 0;
1265
+ max-width: none;
1266
+ }
1267
+ }
1268
+
1269
+ /* 窄屏:单列堆叠 */
1270
+ @media (max-width: 720px) {
1271
+ .formCol {
1272
+ width: 100%;
1273
+ min-width: 0;
1274
+ max-width: none;
1275
+ }
1276
+
1277
+ .resultCol {
1278
+ min-height: 320px;
1279
+ }
1280
+ }
1281
+
1282
+ /* 提示词增强 */
1283
+ .formSectionRow {
1284
+ display: flex;
1285
+ align-items: flex-end;
1286
+ justify-content: space-between;
1287
+ gap: 8px;
1288
+ }
1289
+
1290
+ .formSectionRow .formSection {
1291
+ flex: 1;
1292
+ }
1293
+
1294
+ .enhanceCard {
1295
+ border: 1px solid var(--dsw-alias-border-l1, #e5e7eb);
1296
+ border-radius: 10px;
1297
+ padding: 10px;
1298
+ background: var(--dsw-alias-bg-layer-2, #fafafa);
1299
+ display: flex;
1300
+ flex-direction: column;
1301
+ gap: 8px;
1302
+ }
1303
+
1304
+ .enhanceCardHead {
1305
+ display: flex;
1306
+ align-items: center;
1307
+ justify-content: space-between;
1308
+ gap: 8px;
1309
+ font-size: 12px;
1310
+ color: var(--dsw-alias-label-secondary, #6b7280);
1311
+ }
1312
+
1313
+ .enhanceActions {
1314
+ display: flex;
1315
+ gap: 6px;
1316
+ }
@@ -230,6 +230,9 @@ export function StudioView(props: {
230
230
  const [bitrate, setBitrate] = useState('')
231
231
  const [audioChannel, setAudioChannel] = useState('')
232
232
  const [subtitle, setSubtitle] = useState(false)
233
+ // 提示词增强
234
+ const [enhancing, setEnhancing] = useState(false)
235
+ const [enhancePreview, setEnhancePreview] = useState<string | null>(null)
233
236
  // Stable Audio 参数(仅 Stability 渠道显示)
234
237
  const [seed, setSeed] = useState('')
235
238
  const [steps, setSteps] = useState('')
@@ -578,6 +581,28 @@ export function StudioView(props: {
578
581
  props.showToast('已恢复该次生成的配置,可直接再次生成')
579
582
  }
580
583
 
584
+ /** 调用宿主增强(Agent 默认模型),结果先预览再应用。 */
585
+ const runEnhance = async (): Promise<void> => {
586
+ if (prompt.trim() === '') {
587
+ setError('请先输入文本/提示词,再点击增强')
588
+ return
589
+ }
590
+ setEnhancing(true)
591
+ setError(null)
592
+ try {
593
+ const result = await api.enhancePrompt(prompt.trim(), mode)
594
+ if (result.ok !== true || result.enhanced === undefined) {
595
+ setError(result.message ?? '增强失败,请稍后重试')
596
+ return
597
+ }
598
+ setEnhancePreview(result.enhanced)
599
+ } catch (err) {
600
+ setError(err instanceof Error ? err.message : String(err))
601
+ } finally {
602
+ setEnhancing(false)
603
+ }
604
+ }
605
+
581
606
  /** 删除历史记录(对比任务卡删除该任务的全部模型条目)。 */
582
607
  const deleteHistoryEntries = async (ids: string[]): Promise<void> => {
583
608
  try {
@@ -870,11 +895,29 @@ export function StudioView(props: {
870
895
  ))}
871
896
  </div>
872
897
 
873
- <p className={css.formSection}>输入</p>
898
+ <div className={css.formSectionRow}>
899
+ <p className={css.formSection}>输入</p>
900
+ <button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>
901
+ {enhancing ? '增强中…' : '✨ 增强提示词'}
902
+ </button>
903
+ </div>
874
904
  <label className={css.label}>
875
905
  <span>{mode === 'voice_design' ? '音色描述' : mode === 'tts' ? '文本' : '提示词'}</span>
876
906
  <textarea className={css.textarea} value={prompt} onChange={event => setPrompt(event.target.value)} placeholder={tt('prompt.placeholder')} />
877
907
  </label>
908
+ {enhancePreview !== null ? (
909
+ <div className={css.enhanceCard}>
910
+ <div className={css.enhanceCardHead}>
911
+ <strong>增强结果({modeLabelOf(mode)})</strong>
912
+ <span className={css.enhanceActions}>
913
+ <button type="button" className={css.ghostButton} onClick={() => { setPrompt(enhancePreview); setEnhancePreview(null); props.showToast('已应用增强结果') }}>应用</button>
914
+ <button type="button" className={css.ghostButton} disabled={enhancing} onClick={() => void runEnhance()}>重新生成</button>
915
+ <button type="button" className={css.ghostButton} onClick={() => setEnhancePreview(null)}>放弃</button>
916
+ </span>
917
+ </div>
918
+ <textarea className={css.textarea} value={enhancePreview} readOnly />
919
+ </div>
920
+ ) : null}
878
921
 
879
922
  {mode === 'voice_design' ? (
880
923
  <>
package/src/index.ts CHANGED
@@ -16,8 +16,10 @@ import z from 'schemastery'
16
16
  import type {} from '@deepseek-ai/dsh-host-webserver'
17
17
  import type {} from '@deepseek-ai/dsh-system-prompt'
18
18
  import type {} from '@deepseek-ai/dsh-tools'
19
- import { AUDIOGEN_SETTINGS_NAMESPACE, type ChannelConfig, type ModelMapping } from './protocol.ts'
19
+ import { AUDIOGEN_SETTINGS_NAMESPACE, type AudioMode, type ChannelConfig, type ModelMapping } from './protocol.ts'
20
20
  import { createGenerationBudget } from './audio-scheduler.ts'
21
+ import { enhancePromptText } from './prompt-enhance.ts'
22
+ import { AudioGenError } from './audio-engine.ts'
21
23
  import { makeRoutes, type ChannelsView, type SettingsSeam } from './routes.ts'
22
24
  import type { AudioChannel } from './audio-engine.ts'
23
25
  import { registerAgentAudioTools, type AgentAudioToolConfig } from './agent-audio-tools.ts'
@@ -196,6 +198,13 @@ export function apply(ctx: Context, config?: Config): void {
196
198
  // 全局并发闸门:所有上游调用(面板路由 + Agent 工具)共享「最大并发生成数」。
197
199
  const budget = createGenerationBudget(() => resolve().maxConcurrentGenerations)
198
200
 
201
+ // 提示词增强:复用 Agent 默认模型(agent-default-model 设置),面板与 Agent 工具共用。
202
+ const enhance = async (prompt: string, mode: AudioMode): Promise<string> => {
203
+ const seam = ctx.get('settings') as unknown as SettingsSeam
204
+ if (seam?.describe === undefined) throw new AudioGenError('设置服务不可用,无法增强提示词', 'settings-unavailable')
205
+ return enhancePromptText({ settings: seam, llm: () => ctx.get('llm') }, prompt, mode)
206
+ }
207
+
199
208
  const channelsView = (): ChannelsView => {
200
209
  const value = resolve()
201
210
  return { channels: value.channels, defaultChannelId: value.defaultChannelId }
@@ -209,6 +218,7 @@ export function apply(ctx: Context, config?: Config): void {
209
218
  resolveChannels: channelsView,
210
219
  autoSave: () => resolve().autoSaveToLibrary,
211
220
  budget,
221
+ enhance,
212
222
  })
213
223
  const disposers = routes.map(route => ctx.webServer.register(route))
214
224
  return () => { for (const dispose of disposers) dispose() }
@@ -225,6 +235,7 @@ export function apply(ctx: Context, config?: Config): void {
225
235
  defaultChannelId: value.defaultChannelId,
226
236
  autoSaveToLibrary: value.autoSaveToLibrary,
227
237
  budget,
238
+ enhance,
228
239
  }
229
240
  }), 'dsh-audiogen: agent audio tools')
230
241
  })
@@ -0,0 +1,72 @@
1
+ /**
2
+ * 提示词增强:复用 Agent 当前默认模型(设置「模型」里的 provider/model,
3
+ * 即 agent-default-model 命名空间),宿主端发起一次 LLM 调用把用户 prompt
4
+ * 扩展成更适合生成的任务描述。不新增 API key 配置。
5
+ *
6
+ * 面板「✨ 增强提示词」与 generate_audio 工具的 enhance_prompt 都走这里。
7
+ */
8
+
9
+ import type { AudioMode } from './protocol.ts'
10
+ import { AudioGenError } from './audio-engine.ts'
11
+
12
+ /** 按生成模式给出增强指令(系统提示)。 */
13
+ function instructionsFor(mode: AudioMode): string {
14
+ const common = [
15
+ '你是一个音频提示词增强助手。用户给出一个粗略的音频生成需求,',
16
+ '请将其扩写为一段可直接提交给音频生成模型的中文或英文描述。',
17
+ '只输出增强后的描述本身,不要输出任何解释、前后缀、引号或代码块。',
18
+ '保持用户原始意图,不要改变其核心内容;为最终生成的音频服务。',
19
+ '描述控制在 200-600 字左右。',
20
+ ].join('')
21
+ const perMode: Record<AudioMode, string> = {
22
+ tts: '这是文本转语音(TTS)任务:让文本更适合朗读——口语化、自然、带合适的情感标签(如 (laughs)、(whisper)),避免生僻多音字和超长句,可适当补足上下文使语句完整,但不要改写原意。',
23
+ music: '这是音乐生成任务:扩写音乐风格/情绪/乐器/结构/节奏变化/氛围,使用音频模型熟悉的描述词汇(如 cinematic orchestral、lo-fi、bpm、弦乐进出、旋律动机、前中后段结构),如用户未指定可补充风格建议,但保持原方向。',
24
+ sfx: '这是音效生成任务:扩写声音材质、动作过程、空间感、节奏(先轻后重、清脆短促等)、环境氛围,用具体拟声与材质词,避免抽象概括。',
25
+ voice_design: '这是音色设计任务:扩写人声/音色特征——性别年龄、音域、音质(低沉/清亮/沙哑)、语速、情绪性格、适用场景,用可感知的描述,方便语音模型合成。',
26
+ }
27
+ return common + perMode[mode]
28
+ }
29
+
30
+ export interface PromptEnhanceDeps {
31
+ /** DSH 设置 seam(读 agent-default-model)。 */
32
+ settings: { describe(options?: { redactSecrets?: boolean }): Array<{ ns: unknown; value?: unknown }> }
33
+ /** 宿主 LLM 运行时访问器(延迟读取,调用时才获取)。 */
34
+ llm?: () => unknown
35
+ }
36
+
37
+ /** 读取 Agent 默认模型并调用 LLM 增强,返回增强后的文本。 */
38
+ export async function enhancePromptText(deps: PromptEnhanceDeps, prompt: string, mode: AudioMode): Promise<string> {
39
+ const text = prompt.trim()
40
+ if (text === '') throw new AudioGenError('提示词为空,无法增强', 'enhance-empty-prompt')
41
+ const descriptor = (deps.settings.describe({ redactSecrets: true }) ?? []).find(candidate => String(candidate.ns) === 'agent-default-model')
42
+ const value = (descriptor?.value ?? {}) as { provider?: unknown; model?: unknown }
43
+ const provider = typeof value.provider === 'string' && value.provider.trim() !== '' ? value.provider.trim() : ''
44
+ const model = typeof value.model === 'string' && value.model.trim() !== '' ? value.model.trim() : ''
45
+ if (provider === '' || model === '') {
46
+ throw new AudioGenError('未找到 Agent 默认模型(agent-default-model):请先在「设置 → 模型」中配置默认模型', 'no-default-model')
47
+ }
48
+ const runtime = deps.llm?.() as { stream?: (options: unknown) => AsyncIterable<unknown> } | undefined
49
+ if (runtime === undefined || runtime.stream === undefined) {
50
+ throw new AudioGenError('宿主 LLM 服务不可用(ctx.llm 未注册)', 'llm-unavailable')
51
+ }
52
+ let output = ''
53
+ for await (const chunk of runtime.stream({
54
+ provider,
55
+ model,
56
+ messages: [{ role: 'user', content: text }],
57
+ system: instructionsFor(mode),
58
+ temperature: 0.7,
59
+ maxTokens: 1200,
60
+ })) {
61
+ const record = chunk as { type?: string; text?: string }
62
+ if (record.type === 'text-delta' && typeof record.text === 'string') output += record.text
63
+ }
64
+ return stripFences(output.trim())
65
+ }
66
+
67
+ /** 去掉模型可能包裹的 ``` 代码围栏。 */
68
+ function stripFences(value: string): string {
69
+ if (value === '') return value
70
+ const withoutFence = value.replace(/^```[a-zA-Z]*\s*\n?/, '').replace(/\n?```\s*$/, '')
71
+ return withoutFence.trim()
72
+ }
package/src/protocol.ts CHANGED
@@ -8,7 +8,7 @@
8
8
  export const AUDIOGEN_SETTINGS_NAMESPACE = 'dsh-audiogen'
9
9
 
10
10
  /** Published package version shared by the host updater and the client UI. */
11
- export const PLUGIN_VERSION = '0.4.4'
11
+ export const PLUGIN_VERSION = '0.4.5'
12
12
 
13
13
  /** Same-origin route family (loopback-only, mirroring dsh-imagegen). */
14
14
  export const SETTINGS_API = {
@@ -24,6 +24,9 @@ export const TASK_API = {
24
24
  cancel: '/api/dsh-audiogen/task/cancel',
25
25
  } as const
26
26
 
27
+ /** Loopback-only prompt enhancement route (uses the agent's default model). */
28
+ export const ENHANCE_API = '/api/dsh-audiogen/prompt/enhance' as const
29
+
27
30
  /** Host-mediated built-in provider catalog (channels the user can instantiate). */
28
31
  export const PRESETS_API = '/api/dsh-audiogen/presets' as const
29
32
 
package/src/routes.ts CHANGED
@@ -16,7 +16,7 @@ import { discoverAudioModels } from './audio-models.ts'
16
16
  import { AUDIO_PRESETS } from './audio-presets.ts'
17
17
  import { appendHistory, clearHistory, listHistory, readAudioFile, removeHistory, saveAudioFile, listLibrary, saveToLibrary, updateLibraryEntry, removeLibraryEntries, readLibraryFile } from './audio-store.ts'
18
18
  import {
19
- AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
19
+ AUDIO_API, AUDIOGEN_SETTINGS_NAMESPACE, ENHANCE_API, GENERATE_API, HISTORY_API, LIBRARY_API, MODEL_API, PRESETS_API, SETTINGS_API, TASK_API,
20
20
  LIBRARY_TYPES,
21
21
  type GenerateAudioRequest, type GeneratedAudio, type HistoryEntryInput, type LibraryAudioInput, type LibraryProvenance, type LibraryType,
22
22
  } from './protocol.ts'
@@ -47,6 +47,8 @@ export interface AudiogenRoutesDeps {
47
47
  autoSave: () => boolean
48
48
  /** Global upstream concurrency gate (maxConcurrentGenerations). */
49
49
  budget: GenerationBudget
50
+ /** 提示词增强:调用 Agent 默认模型,返回增强后的文本。 */
51
+ enhance: (prompt: string, mode: GenerateAudioRequest['mode']) => Promise<string>
50
52
  }
51
53
 
52
54
  function isLoopbackRequest(request: IncomingMessage): boolean {
@@ -528,6 +530,27 @@ export function makeRoutes(deps: AudiogenRoutesDeps): WebRoute[] {
528
530
  writeJson(res, 200, { ok: true, aborted: controllers !== undefined ? controllers.size : 0 })
529
531
  },
530
532
  },
533
+ // ------------------------------------------------------- prompt enhance
534
+ {
535
+ kind: 'exact',
536
+ path: ENHANCE_API,
537
+ handler: async (req, res) => {
538
+ if (!guard(req, res, 'POST')) return
539
+ const body = await readJsonBody(req)
540
+ const prompt = typeof body?.prompt === 'string' ? body.prompt.trim() : ''
541
+ if (prompt === '') {
542
+ writeJson(res, 200, { ok: false, code: 'bad-request', message: 'prompt is required' })
543
+ return
544
+ }
545
+ const mode = body?.mode === 'music' ? 'music' : body?.mode === 'sfx' ? 'sfx' : body?.mode === 'voice_design' ? 'voice_design' : 'tts'
546
+ try {
547
+ const enhanced = await deps.enhance(prompt, mode)
548
+ writeJson(res, 200, { ok: true, enhanced })
549
+ } catch (error) {
550
+ writeJson(res, 200, { ok: false, code: 'enhance-failed', message: messageOf(error) })
551
+ }
552
+ },
553
+ },
531
554
  // ----------------------------------------------------------- audio file
532
555
  {
533
556
  kind: 'prefix',