dsh-audiogen 0.4.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/index.js CHANGED
@@ -1,8 +1,10 @@
1
1
  import { SettingsConflictError, installSettingsSection, settingsNamespace } from "@deepseek-ai/dsh-settings";
2
+ import { copyFileSync, existsSync, mkdirSync, readdirSync } from "node:fs";
3
+ import { fileURLToPath } from "node:url";
4
+ import path, { dirname, join } from "node:path";
2
5
  import z from "schemastery";
3
6
  import { randomUUID } from "node:crypto";
4
7
  import { mkdir, readFile, rename, rmdir, unlink, writeFile } from "node:fs/promises";
5
- import path from "node:path";
6
8
  import os from "node:os";
7
9
  import { defineTool } from "@deepseek-ai/dsh-tools";
8
10
  //#region src/protocol.ts
@@ -2215,6 +2217,29 @@ function makeRoutes(deps) {
2215
2217
  }
2216
2218
  //#endregion
2217
2219
  //#region src/agent-audio-tools.ts
2220
+ const audioRefSchema = {
2221
+ type: "object",
2222
+ additionalProperties: false,
2223
+ properties: {
2224
+ id: {
2225
+ type: "string",
2226
+ required: true
2227
+ },
2228
+ url: {
2229
+ type: "string",
2230
+ required: true
2231
+ },
2232
+ mime: {
2233
+ type: "string",
2234
+ required: true
2235
+ },
2236
+ bytes: {
2237
+ type: "integer",
2238
+ required: true
2239
+ },
2240
+ voiceId: { type: "string" }
2241
+ }
2242
+ };
2218
2243
  const resultSchema = {
2219
2244
  type: "object",
2220
2245
  additionalProperties: false,
@@ -2244,34 +2269,35 @@ const resultSchema = {
2244
2269
  audio: {
2245
2270
  type: "array",
2246
2271
  required: true,
2272
+ items: audioRefSchema
2273
+ },
2274
+ resources: {
2275
+ type: "array",
2276
+ items: { type: "string" }
2277
+ },
2278
+ groups: {
2279
+ type: "array",
2247
2280
  items: {
2248
2281
  type: "object",
2249
2282
  additionalProperties: false,
2250
2283
  properties: {
2251
- id: {
2284
+ model: {
2252
2285
  type: "string",
2253
2286
  required: true
2254
2287
  },
2255
- url: {
2256
- type: "string",
2257
- required: true
2258
- },
2259
- mime: {
2260
- type: "string",
2261
- required: true
2288
+ audio: {
2289
+ type: "array",
2290
+ required: true,
2291
+ items: audioRefSchema
2262
2292
  },
2263
- bytes: {
2264
- type: "integer",
2265
- required: true
2293
+ resources: {
2294
+ type: "array",
2295
+ items: { type: "string" }
2266
2296
  },
2267
- voiceId: { type: "string" }
2297
+ error: { type: "string" }
2268
2298
  }
2269
2299
  }
2270
2300
  },
2271
- resources: {
2272
- type: "array",
2273
- items: { type: "string" }
2274
- },
2275
2301
  error: { type: "string" }
2276
2302
  }
2277
2303
  };
@@ -2333,6 +2359,16 @@ function registerAgentAudioTools(ctx, resolve) {
2333
2359
  type: "string",
2334
2360
  description: "One of the configured audio models/voices. Defaults to the first configured model."
2335
2361
  },
2362
+ models: {
2363
+ type: "array",
2364
+ items: { type: "string" },
2365
+ description: "Optional: several configured model aliases to generate the SAME prompt with each one, sequentially, for comparison (e.g. [\"speech-2.8-hd\",\"speech-2.6-hd\"]). Cannot be combined with model; when present, models wins."
2366
+ },
2367
+ model_params: {
2368
+ type: "object",
2369
+ additionalProperties: true,
2370
+ description: "Optional per-model parameter overrides used with \"models\" (automatic by default = all models share the global params). Keys are model aliases; values are partial param objects using the same param names (format, duration, voice, speed, emotion, vol, pitch, sample_rate, bitrate, lyrics, is_instrumental, loop, prompt_influence, seed, steps, cfg_scale, subtitle_enable, aigc_watermark, language_boost, pronunciation_tone, voice_modify, timbre_weights). Unset fields fall back to the global values."
2371
+ },
2336
2372
  voice: {
2337
2373
  type: "string",
2338
2374
  description: "Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio."
@@ -2463,10 +2499,15 @@ function registerAgentAudioTools(ctx, resolve) {
2463
2499
  type: "object",
2464
2500
  additionalProperties: false,
2465
2501
  properties: {
2466
- voice_id: { type: "string" },
2467
- weight: { type: "integer" }
2468
- },
2469
- required: ["voice_id", "weight"]
2502
+ voice_id: {
2503
+ type: "string",
2504
+ required: true
2505
+ },
2506
+ weight: {
2507
+ type: "integer",
2508
+ required: true
2509
+ }
2510
+ }
2470
2511
  },
2471
2512
  description: "MiniMax TTS dual-voice blend weights (timbre_weights)."
2472
2513
  },
@@ -2504,156 +2545,223 @@ function registerAgentAudioTools(ctx, resolve) {
2504
2545
  const config = resolve();
2505
2546
  ensureConfigured(config);
2506
2547
  const mode = args.mode === "music" ? "music" : args.mode === "sfx" ? "sfx" : args.mode === "voice_design" ? "voice_design" : "tts";
2507
- const picked = mode === "voice_design" ? (() => {
2508
- const usable = config.channels.filter((channel) => channel.apiUrl.trim() !== "" && channel.apiKey.trim() !== "");
2509
- const target = usable.find((channel) => channel.id === config.defaultChannelId) ?? usable[0];
2510
- if (target === void 0) throw new AudioGenError("No usable audio channel is configured for voice design.", "no-channel-available");
2548
+ /** 把生成参数(snake_case 入参或 model_params 片段)映射为请求字段。 */
2549
+ const mapParams = (raw) => {
2550
+ const voiceModify = typeof raw.voice_modify === "object" && raw.voice_modify !== null ? (() => {
2551
+ const src = raw.voice_modify;
2552
+ const out = {};
2553
+ if (typeof src.pitch === "number") out.pitch = src.pitch;
2554
+ if (typeof src.intensity === "number") out.intensity = src.intensity;
2555
+ if (typeof src.timbre === "number") out.timbre = src.timbre;
2556
+ if (typeof src.sound_effects === "string" && src.sound_effects.trim() !== "") out.soundEffects = src.sound_effects.trim();
2557
+ return Object.keys(out).length > 0 ? out : void 0;
2558
+ })() : void 0;
2559
+ const timbreWeights = Array.isArray(raw.timbre_weights) ? raw.timbre_weights.filter((item) => typeof item === "object" && item !== null && typeof item.voice_id === "string" && typeof item.weight === "number").map((item) => ({
2560
+ voiceId: item.voice_id.trim(),
2561
+ weight: item.weight
2562
+ })).filter((item) => item.voiceId !== "") : void 0;
2563
+ const stringOrEmpty = (key) => {
2564
+ const value = raw[key];
2565
+ return typeof value === "string" && value.trim() !== "" ? value.trim() : void 0;
2566
+ };
2567
+ const finiteOrUndefined = (key) => {
2568
+ const value = raw[key];
2569
+ return typeof value === "number" && Number.isFinite(value) ? value : void 0;
2570
+ };
2511
2571
  return {
2512
- channel: target,
2513
- alias: "",
2514
- upstream: ""
2572
+ ...stringOrEmpty("voice") !== void 0 ? { voice: stringOrEmpty("voice") } : {},
2573
+ ...stringOrEmpty("preview_text") !== void 0 ? { previewText: stringOrEmpty("preview_text") } : {},
2574
+ ...finiteOrUndefined("speed") !== void 0 ? { speed: finiteOrUndefined("speed") } : {},
2575
+ ...finiteOrUndefined("duration") !== void 0 ? { duration: finiteOrUndefined("duration") } : {},
2576
+ ...stringOrEmpty("lyrics") !== void 0 ? { lyrics: stringOrEmpty("lyrics") } : {},
2577
+ ...typeof raw.is_instrumental === "boolean" ? { isInstrumental: raw.is_instrumental } : {},
2578
+ ...typeof raw.loop === "boolean" ? { loop: raw.loop } : {},
2579
+ ...finiteOrUndefined("prompt_influence") !== void 0 ? { promptInfluence: finiteOrUndefined("prompt_influence") } : {},
2580
+ ...finiteOrUndefined("seed") !== void 0 ? { seed: finiteOrUndefined("seed") } : {},
2581
+ ...finiteOrUndefined("steps") !== void 0 ? { steps: finiteOrUndefined("steps") } : {},
2582
+ ...finiteOrUndefined("cfg_scale") !== void 0 ? { cfgScale: finiteOrUndefined("cfg_scale") } : {},
2583
+ ...stringOrEmpty("format") !== void 0 ? { format: stringOrEmpty("format") } : {},
2584
+ ...stringOrEmpty("emotion") !== void 0 ? { emotion: stringOrEmpty("emotion") } : {},
2585
+ ...finiteOrUndefined("vol") !== void 0 ? { vol: finiteOrUndefined("vol") } : {},
2586
+ ...finiteOrUndefined("pitch") !== void 0 ? { pitch: finiteOrUndefined("pitch") } : {},
2587
+ ...typeof raw.text_normalization === "boolean" ? { textNormalization: raw.text_normalization } : {},
2588
+ ...typeof raw.latex_read === "boolean" ? { latexRead: raw.latex_read } : {},
2589
+ ...Array.isArray(raw.pronunciation_tone) && raw.pronunciation_tone.length > 0 ? { pronunciationTone: raw.pronunciation_tone.filter((item) => typeof item === "string" && item.trim() !== "").map((item) => item.trim()) } : {},
2590
+ ...finiteOrUndefined("sample_rate") !== void 0 ? { sampleRate: finiteOrUndefined("sample_rate") } : {},
2591
+ ...finiteOrUndefined("bitrate") !== void 0 ? { bitrate: finiteOrUndefined("bitrate") } : {},
2592
+ ...finiteOrUndefined("channel") !== void 0 ? { audioChannel: finiteOrUndefined("channel") } : {},
2593
+ ...typeof raw.force_cbr === "boolean" ? { forceCbr: raw.force_cbr } : {},
2594
+ ...typeof raw.subtitle_enable === "boolean" ? { subtitleEnable: raw.subtitle_enable } : {},
2595
+ ...typeof raw.aigc_watermark === "boolean" ? { aigcWatermark: raw.aigc_watermark } : {},
2596
+ ...stringOrEmpty("language_boost") !== void 0 ? { languageBoost: stringOrEmpty("language_boost") } : {},
2597
+ ...voiceModify !== void 0 ? { voiceModify } : {},
2598
+ ...timbreWeights !== void 0 && timbreWeights.length > 0 ? { timbreWeights } : {}
2515
2599
  };
2516
- })() : resolveModel(config, args.model);
2517
- const voiceModify = typeof args.voice_modify === "object" && args.voice_modify !== null ? (() => {
2518
- const raw = args.voice_modify;
2519
- const out = {};
2520
- if (typeof raw.pitch === "number") out.pitch = raw.pitch;
2521
- if (typeof raw.intensity === "number") out.intensity = raw.intensity;
2522
- if (typeof raw.timbre === "number") out.timbre = raw.timbre;
2523
- if (typeof raw.sound_effects === "string" && raw.sound_effects.trim() !== "") out.soundEffects = raw.sound_effects.trim();
2524
- return Object.keys(out).length > 0 ? out : void 0;
2525
- })() : void 0;
2526
- const timbreWeights = Array.isArray(args.timbre_weights) ? args.timbre_weights.filter((item) => typeof item === "object" && item !== null && typeof item.voice_id === "string" && typeof item.weight === "number").map((item) => ({
2527
- voiceId: item.voice_id.trim(),
2528
- weight: item.weight
2529
- })).filter((item) => item.voiceId !== "") : void 0;
2530
- const request = {
2531
- mode,
2532
- model: picked.alias,
2533
- upstream: picked.upstream,
2534
- channelId: picked.channel.id,
2535
- channel: picked.channel.name,
2536
- prompt: args.prompt.trim(),
2537
- ...typeof args.voice === "string" && args.voice.trim() !== "" ? { voice: args.voice.trim() } : {},
2538
- ...typeof args.preview_text === "string" && args.preview_text.trim() !== "" ? { previewText: args.preview_text.trim() } : {},
2539
- ...typeof args.speed === "number" ? { speed: args.speed } : {},
2540
- ...typeof args.duration === "number" ? { duration: args.duration } : {},
2541
- ...typeof args.lyrics === "string" && args.lyrics.trim() !== "" ? { lyrics: args.lyrics.trim() } : {},
2542
- ...typeof args.is_instrumental === "boolean" ? { isInstrumental: args.is_instrumental } : {},
2543
- ...typeof args.loop === "boolean" ? { loop: args.loop } : {},
2544
- ...typeof args.prompt_influence === "number" && Number.isFinite(args.prompt_influence) ? { promptInfluence: args.prompt_influence } : {},
2545
- ...typeof args.seed === "number" && Number.isFinite(args.seed) ? { seed: args.seed } : {},
2546
- ...typeof args.steps === "number" && Number.isFinite(args.steps) ? { steps: args.steps } : {},
2547
- ...typeof args.cfg_scale === "number" && Number.isFinite(args.cfg_scale) ? { cfgScale: args.cfg_scale } : {},
2548
- ...typeof args.format === "string" && args.format.trim() !== "" ? { format: args.format.trim() } : {},
2549
- ...typeof args.emotion === "string" && args.emotion.trim() !== "" ? { emotion: args.emotion.trim() } : {},
2550
- ...typeof args.vol === "number" && Number.isFinite(args.vol) ? { vol: args.vol } : {},
2551
- ...typeof args.pitch === "number" && Number.isFinite(args.pitch) ? { pitch: args.pitch } : {},
2552
- ...typeof args.text_normalization === "boolean" ? { textNormalization: args.text_normalization } : {},
2553
- ...typeof args.latex_read === "boolean" ? { latexRead: args.latex_read } : {},
2554
- ...Array.isArray(args.pronunciation_tone) && args.pronunciation_tone.length > 0 ? { pronunciationTone: args.pronunciation_tone.filter((item) => typeof item === "string" && item.trim() !== "").map((item) => item.trim()) } : {},
2555
- ...typeof args.sample_rate === "number" && Number.isFinite(args.sample_rate) ? { sampleRate: args.sample_rate } : {},
2556
- ...typeof args.bitrate === "number" && Number.isFinite(args.bitrate) ? { bitrate: args.bitrate } : {},
2557
- ...typeof args.channel === "number" && Number.isFinite(args.channel) ? { audioChannel: args.channel } : {},
2558
- ...typeof args.force_cbr === "boolean" ? { forceCbr: args.force_cbr } : {},
2559
- ...typeof args.subtitle_enable === "boolean" ? { subtitleEnable: args.subtitle_enable } : {},
2560
- ...typeof args.aigc_watermark === "boolean" ? { aigcWatermark: args.aigc_watermark } : {},
2561
- ...typeof args.language_boost === "string" && args.language_boost.trim() !== "" ? { languageBoost: args.language_boost.trim() } : {},
2562
- ...voiceModify !== void 0 ? { voiceModify } : {},
2563
- ...timbreWeights !== void 0 && timbreWeights.length > 0 ? { timbreWeights } : {}
2564
2600
  };
2565
- try {
2566
- const outputs = await generateAudio(picked.channel, request, exec.signal);
2567
- const audio = [];
2568
- const saved = [];
2569
- for (const [index, output] of outputs.entries()) {
2570
- const stored = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`);
2571
- saved.push({
2572
- id: stored.id,
2573
- url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
2574
- file: stored.file,
2575
- mime: stored.mime,
2576
- bytes: stored.bytes,
2577
- ...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
2578
- });
2579
- audio.push({
2580
- id: stored.id,
2581
- url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
2582
- mime: stored.mime,
2583
- bytes: stored.bytes,
2584
- ...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
2585
- });
2601
+ const buildRequest = (picked) => {
2602
+ const base = mapParams(args);
2603
+ let override = {};
2604
+ if (typeof args.model_params === "object" && args.model_params !== null) {
2605
+ const perModel = args.model_params[picked.alias];
2606
+ if (typeof perModel === "object" && perModel !== null) override = mapParams(perModel);
2586
2607
  }
2608
+ return {
2609
+ mode,
2610
+ model: picked.alias,
2611
+ upstream: picked.upstream,
2612
+ channelId: picked.channel.id,
2613
+ channel: picked.channel.name,
2614
+ prompt: typeof args.prompt === "string" ? args.prompt.trim() : "",
2615
+ ...base,
2616
+ ...override
2617
+ };
2618
+ };
2619
+ /** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
2620
+ const runOne = async (picked) => {
2621
+ const request = buildRequest(picked);
2587
2622
  try {
2588
- await appendHistory({
2589
- id: randomUUID(),
2590
- createdAt: Date.now(),
2591
- mode: request.mode,
2592
- model: picked.alias,
2593
- prompt: request.prompt,
2594
- ...request.voice === void 0 ? {} : { voice: request.voice },
2595
- ...request.speed === void 0 ? {} : { speed: request.speed },
2596
- ...request.duration === void 0 ? {} : { duration: request.duration },
2597
- ...request.format === void 0 ? {} : { format: request.format },
2598
- audio: outputs.map((output, index) => ({
2599
- id: saved[index].id,
2600
- file: saved[index].file,
2601
- b64: Buffer.from(output.data).toString("base64"),
2602
- mime: saved[index].mime,
2603
- bytes: saved[index].bytes,
2604
- url: saved[index].url,
2623
+ const outputs = await generateAudio(picked.channel, request, exec.signal);
2624
+ const audio = [];
2625
+ const saved = [];
2626
+ for (const [index, output] of outputs.entries()) {
2627
+ const stored = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`);
2628
+ saved.push({
2629
+ id: stored.id,
2630
+ url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
2631
+ file: stored.file,
2632
+ mime: stored.mime,
2633
+ bytes: stored.bytes,
2605
2634
  ...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
2606
- })),
2607
- channelId: picked.channel.id,
2608
- channel: picked.channel.name,
2609
- params: { ...request }
2610
- });
2611
- } catch {}
2612
- const wantSave = args.save_to_library === true || config.autoSaveToLibrary && args.save_to_library !== false;
2613
- let resources;
2614
- if (wantSave) try {
2615
- resources = [(await saveToLibrary({
2616
- audioFiles: saved.map((item) => ({
2617
- id: item.id,
2618
- file: item.file,
2619
- mime: item.mime,
2620
- ...item.voiceId === void 0 ? {} : { voiceId: item.voiceId }
2621
- })),
2622
- type: libraryTypeOf(request.mode, args.library_type),
2623
- ...typeof args.library_name === "string" && args.library_name.trim() !== "" ? { name: args.library_name.trim() } : {},
2624
- ...Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag) => typeof tag === "string" && tag.trim() !== "").map((tag) => tag.trim()) } : {},
2625
- provenance: {
2635
+ });
2636
+ audio.push({
2637
+ id: stored.id,
2638
+ url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
2639
+ mime: stored.mime,
2640
+ bytes: stored.bytes,
2641
+ ...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
2642
+ });
2643
+ }
2644
+ try {
2645
+ await appendHistory({
2646
+ id: randomUUID(),
2647
+ createdAt: Date.now(),
2626
2648
  mode: request.mode,
2627
- prompt: request.prompt,
2628
- channel: picked.channel.name,
2629
- channelId: picked.channel.id,
2630
- apiUrl: picked.channel.apiUrl,
2631
2649
  model: picked.alias,
2632
- upstream: picked.upstream,
2650
+ prompt: request.prompt,
2633
2651
  ...request.voice === void 0 ? {} : { voice: request.voice },
2652
+ ...request.speed === void 0 ? {} : { speed: request.speed },
2653
+ ...request.duration === void 0 ? {} : { duration: request.duration },
2654
+ ...request.format === void 0 ? {} : { format: request.format },
2655
+ audio: outputs.map((output, index) => ({
2656
+ id: saved[index].id,
2657
+ file: saved[index].file,
2658
+ b64: Buffer.from(output.data).toString("base64"),
2659
+ mime: saved[index].mime,
2660
+ bytes: saved[index].bytes,
2661
+ url: saved[index].url,
2662
+ ...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
2663
+ })),
2664
+ channelId: picked.channel.id,
2665
+ channel: picked.channel.name,
2634
2666
  params: { ...request }
2635
- }
2636
- })).id];
2637
- } catch {}
2667
+ });
2668
+ } catch {}
2669
+ const wantSave = args.save_to_library === true || config.autoSaveToLibrary && args.save_to_library !== false;
2670
+ let resources;
2671
+ if (wantSave) try {
2672
+ resources = [(await saveToLibrary({
2673
+ audioFiles: saved.map((item) => ({
2674
+ id: item.id,
2675
+ file: item.file,
2676
+ mime: item.mime,
2677
+ ...item.voiceId === void 0 ? {} : { voiceId: item.voiceId }
2678
+ })),
2679
+ type: libraryTypeOf(request.mode, args.library_type),
2680
+ ...typeof args.library_name === "string" && args.library_name.trim() !== "" ? { name: args.library_name.trim() } : {},
2681
+ ...Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag) => typeof tag === "string" && tag.trim() !== "").map((tag) => tag.trim()) } : {},
2682
+ provenance: {
2683
+ mode: request.mode,
2684
+ prompt: request.prompt,
2685
+ channel: picked.channel.name,
2686
+ channelId: picked.channel.id,
2687
+ apiUrl: picked.channel.apiUrl,
2688
+ model: picked.alias,
2689
+ upstream: picked.upstream,
2690
+ ...request.voice === void 0 ? {} : { voice: request.voice },
2691
+ params: { ...request }
2692
+ }
2693
+ })).id];
2694
+ } catch {}
2695
+ return {
2696
+ model: picked.alias,
2697
+ audio,
2698
+ ...resources === void 0 ? {} : { resources }
2699
+ };
2700
+ } catch (error) {
2701
+ if (exec.signal?.aborted === true) throw error;
2702
+ return {
2703
+ model: picked.alias,
2704
+ audio: [],
2705
+ error: error instanceof Error ? error.message : String(error)
2706
+ };
2707
+ }
2708
+ };
2709
+ const requestedModels = Array.isArray(args.models) ? [...new Set(args.models.filter((item) => typeof item === "string" && item.trim() !== "").map((item) => item.trim()))] : [];
2710
+ if (requestedModels.length > 0 && mode !== "voice_design") {
2711
+ const groups = [];
2712
+ let succeeded = 0;
2713
+ for (const alias of requestedModels) {
2714
+ let picked;
2715
+ try {
2716
+ picked = resolveModel(config, alias);
2717
+ } catch (error) {
2718
+ groups.push({
2719
+ model: alias,
2720
+ audio: [],
2721
+ error: error instanceof Error ? error.message : String(error)
2722
+ });
2723
+ continue;
2724
+ }
2725
+ const group = await runOne(picked);
2726
+ groups.push(group);
2727
+ if (group.error === void 0) succeeded++;
2728
+ }
2638
2729
  return {
2639
- status: "completed",
2640
- message: "Audio generation completed. The audio files can be played/downloaded from the returned URLs.",
2641
- mode: request.mode,
2642
- model: picked.alias,
2643
- audio,
2644
- ...resources === void 0 ? {} : { resources }
2730
+ status: succeeded > 0 ? "completed" : "failed",
2731
+ message: succeeded > 0 ? `Generated ${succeeded}/${groups.length} model(s) with the same prompt for comparison. The audio files can be played/downloaded from the returned URLs.` : "All model generations failed.",
2732
+ mode,
2733
+ model: groups[0]?.model ?? requestedModels[0],
2734
+ audio: groups.flatMap((group) => group.audio),
2735
+ groups,
2736
+ ...succeeded === 0 ? { error: groups.map((group) => `${group.model}: ${group.error ?? ""}`).filter((item) => !item.endsWith(": ")).join(";") } : {}
2645
2737
  };
2646
- } catch (error) {
2647
- if (exec.signal?.aborted === true) throw error;
2738
+ }
2739
+ const one = await runOne(mode === "voice_design" ? (() => {
2740
+ const usable = config.channels.filter((channel) => channel.apiUrl.trim() !== "" && channel.apiKey.trim() !== "");
2741
+ const target = usable.find((channel) => channel.id === config.defaultChannelId) ?? usable[0];
2742
+ if (target === void 0) throw new AudioGenError("No usable audio channel is configured for voice design.", "no-channel-available");
2648
2743
  return {
2649
- status: "failed",
2650
- message: "Audio generation failed.",
2651
- mode: request.mode,
2652
- model: picked.alias,
2653
- audio: [],
2654
- error: error instanceof Error ? error.message : String(error)
2744
+ channel: target,
2745
+ alias: "",
2746
+ upstream: ""
2655
2747
  };
2656
- }
2748
+ })() : resolveModel(config, args.model));
2749
+ if (one.error !== void 0) return {
2750
+ status: "failed",
2751
+ message: "Audio generation failed.",
2752
+ mode,
2753
+ model: one.model,
2754
+ audio: [],
2755
+ error: one.error
2756
+ };
2757
+ return {
2758
+ status: "completed",
2759
+ message: "Audio generation completed. The audio files can be played/downloaded from the returned URLs.",
2760
+ mode,
2761
+ model: one.model,
2762
+ audio: one.audio,
2763
+ ...one.resources === void 0 ? {} : { resources: one.resources }
2764
+ };
2657
2765
  }
2658
2766
  }));
2659
2767
  const searchDisposer = ctx.tools.register(defineTool({
@@ -2831,6 +2939,29 @@ function guidanceFor(channels, defaultChannelId) {
2831
2939
  }).join(";");
2832
2940
  return `${AUDIOGEN_GUIDANCE} 当前渠道与模型:${table}。`;
2833
2941
  }
2942
+ /**
2943
+ * 把随包分发的技能(skills/<id>/SKILL.md,含 frontmatter)同步到 DSH 用户技能根
2944
+ * `~/.dsh/skills/<id>/SKILL.md` —— DSH web 会话的 skill-filesystem(standard 等
2945
+ * preset 行)会扫描用户根,使会话可直接触发这些技能。仅创建缺失文件,绝不覆盖
2946
+ * 用户已有内容;任何失败仅告警,不影响插件本身。
2947
+ */
2948
+ function syncBundledSkills() {
2949
+ try {
2950
+ const sourceRoot = join(dirname(dirname(fileURLToPath(import.meta.url))), "skills");
2951
+ if (existsSync(sourceRoot) !== true) return;
2952
+ const targetRoot = join(process.env.DSH_HOME ?? join(process.env.HOME ?? "", ".dsh"), "skills");
2953
+ for (const entry of readdirSync(sourceRoot, { withFileTypes: true })) {
2954
+ if (entry.isDirectory() !== true) continue;
2955
+ const sourceFile = join(sourceRoot, entry.name, "SKILL.md");
2956
+ if (existsSync(sourceFile) !== true) continue;
2957
+ const targetDir = join(targetRoot, entry.name);
2958
+ const targetFile = join(targetDir, "SKILL.md");
2959
+ if (existsSync(targetFile)) continue;
2960
+ mkdirSync(targetDir, { recursive: true });
2961
+ copyFileSync(sourceFile, targetFile);
2962
+ }
2963
+ } catch {}
2964
+ }
2834
2965
  function normalizeChannels(value) {
2835
2966
  if (!Array.isArray(value)) return [];
2836
2967
  const out = [];
@@ -2862,6 +2993,7 @@ function normalizeChannels(value) {
2862
2993
  return out;
2863
2994
  }
2864
2995
  function apply(ctx, config) {
2996
+ syncBundledSkills();
2865
2997
  let current = () => config ?? {};
2866
2998
  const resolve = () => {
2867
2999
  const value = current() ?? {};
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "dsh-audiogen",
3
3
  "description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
4
- "version": "0.4.0",
4
+ "version": "0.4.1",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",
7
7
  "exports": {
@@ -1,3 +1,8 @@
1
+ ---
2
+ name: dsh-audiogen-voice-design
3
+ description: DSH AI 音频插件(dsh-audiogen)的音色设计技能:调用 generate_audio(mode=voice_design) 并指定厂商/渠道——MiniMax(POST /v1/voice_design,prompt + preview_text)或 ElevenLabs(POST /v1/text-to-voice/design,voice_description + 试听文本 100-1000 字符,过短自动生成;返回 previews[].audio_base_64 与 generated_voice_id 供后续 TTS 复用)。
4
+ whenToUse: 用户请求设计/创建新音色、音色试听,或触发 /audio:design 时使用。
5
+ ---
1
6
  # 音色/音效设计
2
7
 
3
8
  ## 触发
@@ -1,3 +1,8 @@
1
+ ---
2
+ name: dsh-audiogen-music
3
+ description: DSH AI 音频插件(dsh-audiogen)的音乐生成技能:调用 generate_audio(mode=music);覆盖 MiniMax(music-3.0/2.6/cover,lyrics 歌词、is_instrumental 纯音乐、audio_setting 采样率 16000-44100/码率 32000-256000/格式 mp3-wav-pcm、时长)、ElevenLabs(/v1/music,music_v2,时长 3-600s、lyrics_text、force_instrumental)与 Stability(stable-audio 2/2.5/3,官方 v2beta 或 OpenAI 兼容 /v1/audio/speech 双通道,seed/steps/cfg_scale/duration)。
4
+ whenToUse: 用户请求生成音乐、配乐、BGM、纯音乐、歌曲,或触发 /audio:music 时使用;MiniMax 未给歌词且未要求纯音乐时,先补一段歌词或设置 is_instrumental。
5
+ ---
1
6
  # 音乐生成
2
7
 
3
8
  ## 触发
@@ -1,3 +1,8 @@
1
+ ---
2
+ name: dsh-audiogen-sfx
3
+ description: DSH AI 音频插件(dsh-audiogen)的音效生成技能:调用 generate_audio(mode=sfx);覆盖 ElevenLabs(/v1/sound-generation,eleven_text_to_sound_v2,loop 无缝循环、prompt_influence 0-1、duration_seconds 0.5-30)与 MiniMax、Stability 等渠道的对应字段与常见错误处理。
4
+ whenToUse: 用户请求生成音效、提示音、环境音、UI 音,或触发 /audio:sfx 时使用。
5
+ ---
1
6
  # 音效生成
2
7
 
3
8
  ## 触发
@@ -1,3 +1,8 @@
1
+ ---
2
+ name: dsh-audiogen-tts
3
+ description: DSH AI 音频插件(dsh-audiogen)的 TTS 文本转语音技能:先确认渠道/模型/音色,再调用 generate_audio(mode=tts);包含 MiniMax 官方 t2a_v2 全字段(语速/音量/音调/情绪/采样率/码率/声道/发音词典/字幕/变声/双音色混合)、ElevenLabs 与 Stable Audio 的对应参数说明,以及常见错误(voice-required、网关 404 Invalid URL 等)的处理。
4
+ whenToUse: 用户提出朗读、配音、语音合成、TTS,或触发 /audio:tts 时使用;MiniMax 必须提供音色 voice_id。
5
+ ---
1
6
  # TTS 文本转语音
2
7
 
3
8
  ## 触发