dsh-audiogen 0.4.0 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +665 -407
- package/lib/client.js.map +1 -1
- package/lib/index.js +289 -157
- package/package.json +1 -1
- package/skills/design/SKILL.md +5 -0
- package/skills/music/SKILL.md +5 -0
- package/skills/sfx/SKILL.md +5 -0
- package/skills/tts/SKILL.md +5 -0
- package/src/agent-audio-tools.ts +255 -149
- package/src/client/audio-panel.module.css +87 -0
- package/src/client/studio-view.tsx +270 -91
- package/src/index.ts +32 -0
- package/src/protocol.ts +1 -1
package/lib/index.js
CHANGED
|
@@ -1,8 +1,10 @@
|
|
|
1
1
|
import { SettingsConflictError, installSettingsSection, settingsNamespace } from "@deepseek-ai/dsh-settings";
|
|
2
|
+
import { copyFileSync, existsSync, mkdirSync, readdirSync } from "node:fs";
|
|
3
|
+
import { fileURLToPath } from "node:url";
|
|
4
|
+
import path, { dirname, join } from "node:path";
|
|
2
5
|
import z from "schemastery";
|
|
3
6
|
import { randomUUID } from "node:crypto";
|
|
4
7
|
import { mkdir, readFile, rename, rmdir, unlink, writeFile } from "node:fs/promises";
|
|
5
|
-
import path from "node:path";
|
|
6
8
|
import os from "node:os";
|
|
7
9
|
import { defineTool } from "@deepseek-ai/dsh-tools";
|
|
8
10
|
//#region src/protocol.ts
|
|
@@ -2215,6 +2217,29 @@ function makeRoutes(deps) {
|
|
|
2215
2217
|
}
|
|
2216
2218
|
//#endregion
|
|
2217
2219
|
//#region src/agent-audio-tools.ts
|
|
2220
|
+
const audioRefSchema = {
|
|
2221
|
+
type: "object",
|
|
2222
|
+
additionalProperties: false,
|
|
2223
|
+
properties: {
|
|
2224
|
+
id: {
|
|
2225
|
+
type: "string",
|
|
2226
|
+
required: true
|
|
2227
|
+
},
|
|
2228
|
+
url: {
|
|
2229
|
+
type: "string",
|
|
2230
|
+
required: true
|
|
2231
|
+
},
|
|
2232
|
+
mime: {
|
|
2233
|
+
type: "string",
|
|
2234
|
+
required: true
|
|
2235
|
+
},
|
|
2236
|
+
bytes: {
|
|
2237
|
+
type: "integer",
|
|
2238
|
+
required: true
|
|
2239
|
+
},
|
|
2240
|
+
voiceId: { type: "string" }
|
|
2241
|
+
}
|
|
2242
|
+
};
|
|
2218
2243
|
const resultSchema = {
|
|
2219
2244
|
type: "object",
|
|
2220
2245
|
additionalProperties: false,
|
|
@@ -2244,34 +2269,35 @@ const resultSchema = {
|
|
|
2244
2269
|
audio: {
|
|
2245
2270
|
type: "array",
|
|
2246
2271
|
required: true,
|
|
2272
|
+
items: audioRefSchema
|
|
2273
|
+
},
|
|
2274
|
+
resources: {
|
|
2275
|
+
type: "array",
|
|
2276
|
+
items: { type: "string" }
|
|
2277
|
+
},
|
|
2278
|
+
groups: {
|
|
2279
|
+
type: "array",
|
|
2247
2280
|
items: {
|
|
2248
2281
|
type: "object",
|
|
2249
2282
|
additionalProperties: false,
|
|
2250
2283
|
properties: {
|
|
2251
|
-
|
|
2284
|
+
model: {
|
|
2252
2285
|
type: "string",
|
|
2253
2286
|
required: true
|
|
2254
2287
|
},
|
|
2255
|
-
|
|
2256
|
-
type: "
|
|
2257
|
-
required: true
|
|
2258
|
-
|
|
2259
|
-
mime: {
|
|
2260
|
-
type: "string",
|
|
2261
|
-
required: true
|
|
2288
|
+
audio: {
|
|
2289
|
+
type: "array",
|
|
2290
|
+
required: true,
|
|
2291
|
+
items: audioRefSchema
|
|
2262
2292
|
},
|
|
2263
|
-
|
|
2264
|
-
type: "
|
|
2265
|
-
|
|
2293
|
+
resources: {
|
|
2294
|
+
type: "array",
|
|
2295
|
+
items: { type: "string" }
|
|
2266
2296
|
},
|
|
2267
|
-
|
|
2297
|
+
error: { type: "string" }
|
|
2268
2298
|
}
|
|
2269
2299
|
}
|
|
2270
2300
|
},
|
|
2271
|
-
resources: {
|
|
2272
|
-
type: "array",
|
|
2273
|
-
items: { type: "string" }
|
|
2274
|
-
},
|
|
2275
2301
|
error: { type: "string" }
|
|
2276
2302
|
}
|
|
2277
2303
|
};
|
|
@@ -2333,6 +2359,16 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
2333
2359
|
type: "string",
|
|
2334
2360
|
description: "One of the configured audio models/voices. Defaults to the first configured model."
|
|
2335
2361
|
},
|
|
2362
|
+
models: {
|
|
2363
|
+
type: "array",
|
|
2364
|
+
items: { type: "string" },
|
|
2365
|
+
description: "Optional: several configured model aliases to generate the SAME prompt with each one, sequentially, for comparison (e.g. [\"speech-2.8-hd\",\"speech-2.6-hd\"]). Cannot be combined with model; when present, models wins."
|
|
2366
|
+
},
|
|
2367
|
+
model_params: {
|
|
2368
|
+
type: "object",
|
|
2369
|
+
additionalProperties: true,
|
|
2370
|
+
description: "Optional per-model parameter overrides used with \"models\" (automatic by default = all models share the global params). Keys are model aliases; values are partial param objects using the same param names (format, duration, voice, speed, emotion, vol, pitch, sample_rate, bitrate, lyrics, is_instrumental, loop, prompt_influence, seed, steps, cfg_scale, subtitle_enable, aigc_watermark, language_boost, pronunciation_tone, voice_modify, timbre_weights). Unset fields fall back to the global values."
|
|
2371
|
+
},
|
|
2336
2372
|
voice: {
|
|
2337
2373
|
type: "string",
|
|
2338
2374
|
description: "Optional voice id/name for TTS providers. Required for MiniMax TTS (e.g. male-qn-qingse, female-shaonv); fetch the account voices in Settings > Plugins > AI Audio."
|
|
@@ -2463,10 +2499,15 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
2463
2499
|
type: "object",
|
|
2464
2500
|
additionalProperties: false,
|
|
2465
2501
|
properties: {
|
|
2466
|
-
voice_id: {
|
|
2467
|
-
|
|
2468
|
-
|
|
2469
|
-
|
|
2502
|
+
voice_id: {
|
|
2503
|
+
type: "string",
|
|
2504
|
+
required: true
|
|
2505
|
+
},
|
|
2506
|
+
weight: {
|
|
2507
|
+
type: "integer",
|
|
2508
|
+
required: true
|
|
2509
|
+
}
|
|
2510
|
+
}
|
|
2470
2511
|
},
|
|
2471
2512
|
description: "MiniMax TTS dual-voice blend weights (timbre_weights)."
|
|
2472
2513
|
},
|
|
@@ -2504,156 +2545,223 @@ function registerAgentAudioTools(ctx, resolve) {
|
|
|
2504
2545
|
const config = resolve();
|
|
2505
2546
|
ensureConfigured(config);
|
|
2506
2547
|
const mode = args.mode === "music" ? "music" : args.mode === "sfx" ? "sfx" : args.mode === "voice_design" ? "voice_design" : "tts";
|
|
2507
|
-
|
|
2508
|
-
|
|
2509
|
-
const
|
|
2510
|
-
|
|
2548
|
+
/** 把生成参数(snake_case 入参或 model_params 片段)映射为请求字段。 */
|
|
2549
|
+
const mapParams = (raw) => {
|
|
2550
|
+
const voiceModify = typeof raw.voice_modify === "object" && raw.voice_modify !== null ? (() => {
|
|
2551
|
+
const src = raw.voice_modify;
|
|
2552
|
+
const out = {};
|
|
2553
|
+
if (typeof src.pitch === "number") out.pitch = src.pitch;
|
|
2554
|
+
if (typeof src.intensity === "number") out.intensity = src.intensity;
|
|
2555
|
+
if (typeof src.timbre === "number") out.timbre = src.timbre;
|
|
2556
|
+
if (typeof src.sound_effects === "string" && src.sound_effects.trim() !== "") out.soundEffects = src.sound_effects.trim();
|
|
2557
|
+
return Object.keys(out).length > 0 ? out : void 0;
|
|
2558
|
+
})() : void 0;
|
|
2559
|
+
const timbreWeights = Array.isArray(raw.timbre_weights) ? raw.timbre_weights.filter((item) => typeof item === "object" && item !== null && typeof item.voice_id === "string" && typeof item.weight === "number").map((item) => ({
|
|
2560
|
+
voiceId: item.voice_id.trim(),
|
|
2561
|
+
weight: item.weight
|
|
2562
|
+
})).filter((item) => item.voiceId !== "") : void 0;
|
|
2563
|
+
const stringOrEmpty = (key) => {
|
|
2564
|
+
const value = raw[key];
|
|
2565
|
+
return typeof value === "string" && value.trim() !== "" ? value.trim() : void 0;
|
|
2566
|
+
};
|
|
2567
|
+
const finiteOrUndefined = (key) => {
|
|
2568
|
+
const value = raw[key];
|
|
2569
|
+
return typeof value === "number" && Number.isFinite(value) ? value : void 0;
|
|
2570
|
+
};
|
|
2511
2571
|
return {
|
|
2512
|
-
|
|
2513
|
-
|
|
2514
|
-
|
|
2572
|
+
...stringOrEmpty("voice") !== void 0 ? { voice: stringOrEmpty("voice") } : {},
|
|
2573
|
+
...stringOrEmpty("preview_text") !== void 0 ? { previewText: stringOrEmpty("preview_text") } : {},
|
|
2574
|
+
...finiteOrUndefined("speed") !== void 0 ? { speed: finiteOrUndefined("speed") } : {},
|
|
2575
|
+
...finiteOrUndefined("duration") !== void 0 ? { duration: finiteOrUndefined("duration") } : {},
|
|
2576
|
+
...stringOrEmpty("lyrics") !== void 0 ? { lyrics: stringOrEmpty("lyrics") } : {},
|
|
2577
|
+
...typeof raw.is_instrumental === "boolean" ? { isInstrumental: raw.is_instrumental } : {},
|
|
2578
|
+
...typeof raw.loop === "boolean" ? { loop: raw.loop } : {},
|
|
2579
|
+
...finiteOrUndefined("prompt_influence") !== void 0 ? { promptInfluence: finiteOrUndefined("prompt_influence") } : {},
|
|
2580
|
+
...finiteOrUndefined("seed") !== void 0 ? { seed: finiteOrUndefined("seed") } : {},
|
|
2581
|
+
...finiteOrUndefined("steps") !== void 0 ? { steps: finiteOrUndefined("steps") } : {},
|
|
2582
|
+
...finiteOrUndefined("cfg_scale") !== void 0 ? { cfgScale: finiteOrUndefined("cfg_scale") } : {},
|
|
2583
|
+
...stringOrEmpty("format") !== void 0 ? { format: stringOrEmpty("format") } : {},
|
|
2584
|
+
...stringOrEmpty("emotion") !== void 0 ? { emotion: stringOrEmpty("emotion") } : {},
|
|
2585
|
+
...finiteOrUndefined("vol") !== void 0 ? { vol: finiteOrUndefined("vol") } : {},
|
|
2586
|
+
...finiteOrUndefined("pitch") !== void 0 ? { pitch: finiteOrUndefined("pitch") } : {},
|
|
2587
|
+
...typeof raw.text_normalization === "boolean" ? { textNormalization: raw.text_normalization } : {},
|
|
2588
|
+
...typeof raw.latex_read === "boolean" ? { latexRead: raw.latex_read } : {},
|
|
2589
|
+
...Array.isArray(raw.pronunciation_tone) && raw.pronunciation_tone.length > 0 ? { pronunciationTone: raw.pronunciation_tone.filter((item) => typeof item === "string" && item.trim() !== "").map((item) => item.trim()) } : {},
|
|
2590
|
+
...finiteOrUndefined("sample_rate") !== void 0 ? { sampleRate: finiteOrUndefined("sample_rate") } : {},
|
|
2591
|
+
...finiteOrUndefined("bitrate") !== void 0 ? { bitrate: finiteOrUndefined("bitrate") } : {},
|
|
2592
|
+
...finiteOrUndefined("channel") !== void 0 ? { audioChannel: finiteOrUndefined("channel") } : {},
|
|
2593
|
+
...typeof raw.force_cbr === "boolean" ? { forceCbr: raw.force_cbr } : {},
|
|
2594
|
+
...typeof raw.subtitle_enable === "boolean" ? { subtitleEnable: raw.subtitle_enable } : {},
|
|
2595
|
+
...typeof raw.aigc_watermark === "boolean" ? { aigcWatermark: raw.aigc_watermark } : {},
|
|
2596
|
+
...stringOrEmpty("language_boost") !== void 0 ? { languageBoost: stringOrEmpty("language_boost") } : {},
|
|
2597
|
+
...voiceModify !== void 0 ? { voiceModify } : {},
|
|
2598
|
+
...timbreWeights !== void 0 && timbreWeights.length > 0 ? { timbreWeights } : {}
|
|
2515
2599
|
};
|
|
2516
|
-
})() : resolveModel(config, args.model);
|
|
2517
|
-
const voiceModify = typeof args.voice_modify === "object" && args.voice_modify !== null ? (() => {
|
|
2518
|
-
const raw = args.voice_modify;
|
|
2519
|
-
const out = {};
|
|
2520
|
-
if (typeof raw.pitch === "number") out.pitch = raw.pitch;
|
|
2521
|
-
if (typeof raw.intensity === "number") out.intensity = raw.intensity;
|
|
2522
|
-
if (typeof raw.timbre === "number") out.timbre = raw.timbre;
|
|
2523
|
-
if (typeof raw.sound_effects === "string" && raw.sound_effects.trim() !== "") out.soundEffects = raw.sound_effects.trim();
|
|
2524
|
-
return Object.keys(out).length > 0 ? out : void 0;
|
|
2525
|
-
})() : void 0;
|
|
2526
|
-
const timbreWeights = Array.isArray(args.timbre_weights) ? args.timbre_weights.filter((item) => typeof item === "object" && item !== null && typeof item.voice_id === "string" && typeof item.weight === "number").map((item) => ({
|
|
2527
|
-
voiceId: item.voice_id.trim(),
|
|
2528
|
-
weight: item.weight
|
|
2529
|
-
})).filter((item) => item.voiceId !== "") : void 0;
|
|
2530
|
-
const request = {
|
|
2531
|
-
mode,
|
|
2532
|
-
model: picked.alias,
|
|
2533
|
-
upstream: picked.upstream,
|
|
2534
|
-
channelId: picked.channel.id,
|
|
2535
|
-
channel: picked.channel.name,
|
|
2536
|
-
prompt: args.prompt.trim(),
|
|
2537
|
-
...typeof args.voice === "string" && args.voice.trim() !== "" ? { voice: args.voice.trim() } : {},
|
|
2538
|
-
...typeof args.preview_text === "string" && args.preview_text.trim() !== "" ? { previewText: args.preview_text.trim() } : {},
|
|
2539
|
-
...typeof args.speed === "number" ? { speed: args.speed } : {},
|
|
2540
|
-
...typeof args.duration === "number" ? { duration: args.duration } : {},
|
|
2541
|
-
...typeof args.lyrics === "string" && args.lyrics.trim() !== "" ? { lyrics: args.lyrics.trim() } : {},
|
|
2542
|
-
...typeof args.is_instrumental === "boolean" ? { isInstrumental: args.is_instrumental } : {},
|
|
2543
|
-
...typeof args.loop === "boolean" ? { loop: args.loop } : {},
|
|
2544
|
-
...typeof args.prompt_influence === "number" && Number.isFinite(args.prompt_influence) ? { promptInfluence: args.prompt_influence } : {},
|
|
2545
|
-
...typeof args.seed === "number" && Number.isFinite(args.seed) ? { seed: args.seed } : {},
|
|
2546
|
-
...typeof args.steps === "number" && Number.isFinite(args.steps) ? { steps: args.steps } : {},
|
|
2547
|
-
...typeof args.cfg_scale === "number" && Number.isFinite(args.cfg_scale) ? { cfgScale: args.cfg_scale } : {},
|
|
2548
|
-
...typeof args.format === "string" && args.format.trim() !== "" ? { format: args.format.trim() } : {},
|
|
2549
|
-
...typeof args.emotion === "string" && args.emotion.trim() !== "" ? { emotion: args.emotion.trim() } : {},
|
|
2550
|
-
...typeof args.vol === "number" && Number.isFinite(args.vol) ? { vol: args.vol } : {},
|
|
2551
|
-
...typeof args.pitch === "number" && Number.isFinite(args.pitch) ? { pitch: args.pitch } : {},
|
|
2552
|
-
...typeof args.text_normalization === "boolean" ? { textNormalization: args.text_normalization } : {},
|
|
2553
|
-
...typeof args.latex_read === "boolean" ? { latexRead: args.latex_read } : {},
|
|
2554
|
-
...Array.isArray(args.pronunciation_tone) && args.pronunciation_tone.length > 0 ? { pronunciationTone: args.pronunciation_tone.filter((item) => typeof item === "string" && item.trim() !== "").map((item) => item.trim()) } : {},
|
|
2555
|
-
...typeof args.sample_rate === "number" && Number.isFinite(args.sample_rate) ? { sampleRate: args.sample_rate } : {},
|
|
2556
|
-
...typeof args.bitrate === "number" && Number.isFinite(args.bitrate) ? { bitrate: args.bitrate } : {},
|
|
2557
|
-
...typeof args.channel === "number" && Number.isFinite(args.channel) ? { audioChannel: args.channel } : {},
|
|
2558
|
-
...typeof args.force_cbr === "boolean" ? { forceCbr: args.force_cbr } : {},
|
|
2559
|
-
...typeof args.subtitle_enable === "boolean" ? { subtitleEnable: args.subtitle_enable } : {},
|
|
2560
|
-
...typeof args.aigc_watermark === "boolean" ? { aigcWatermark: args.aigc_watermark } : {},
|
|
2561
|
-
...typeof args.language_boost === "string" && args.language_boost.trim() !== "" ? { languageBoost: args.language_boost.trim() } : {},
|
|
2562
|
-
...voiceModify !== void 0 ? { voiceModify } : {},
|
|
2563
|
-
...timbreWeights !== void 0 && timbreWeights.length > 0 ? { timbreWeights } : {}
|
|
2564
2600
|
};
|
|
2565
|
-
|
|
2566
|
-
const
|
|
2567
|
-
|
|
2568
|
-
|
|
2569
|
-
|
|
2570
|
-
|
|
2571
|
-
saved.push({
|
|
2572
|
-
id: stored.id,
|
|
2573
|
-
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
2574
|
-
file: stored.file,
|
|
2575
|
-
mime: stored.mime,
|
|
2576
|
-
bytes: stored.bytes,
|
|
2577
|
-
...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
|
|
2578
|
-
});
|
|
2579
|
-
audio.push({
|
|
2580
|
-
id: stored.id,
|
|
2581
|
-
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
2582
|
-
mime: stored.mime,
|
|
2583
|
-
bytes: stored.bytes,
|
|
2584
|
-
...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
|
|
2585
|
-
});
|
|
2601
|
+
const buildRequest = (picked) => {
|
|
2602
|
+
const base = mapParams(args);
|
|
2603
|
+
let override = {};
|
|
2604
|
+
if (typeof args.model_params === "object" && args.model_params !== null) {
|
|
2605
|
+
const perModel = args.model_params[picked.alias];
|
|
2606
|
+
if (typeof perModel === "object" && perModel !== null) override = mapParams(perModel);
|
|
2586
2607
|
}
|
|
2608
|
+
return {
|
|
2609
|
+
mode,
|
|
2610
|
+
model: picked.alias,
|
|
2611
|
+
upstream: picked.upstream,
|
|
2612
|
+
channelId: picked.channel.id,
|
|
2613
|
+
channel: picked.channel.name,
|
|
2614
|
+
prompt: typeof args.prompt === "string" ? args.prompt.trim() : "",
|
|
2615
|
+
...base,
|
|
2616
|
+
...override
|
|
2617
|
+
};
|
|
2618
|
+
};
|
|
2619
|
+
/** 单模型执行:生成 + 保存文件 + 历史 + 可选资源库;错误收敛为分组结果。 */
|
|
2620
|
+
const runOne = async (picked) => {
|
|
2621
|
+
const request = buildRequest(picked);
|
|
2587
2622
|
try {
|
|
2588
|
-
await
|
|
2589
|
-
|
|
2590
|
-
|
|
2591
|
-
|
|
2592
|
-
|
|
2593
|
-
|
|
2594
|
-
|
|
2595
|
-
|
|
2596
|
-
|
|
2597
|
-
|
|
2598
|
-
|
|
2599
|
-
id: saved[index].id,
|
|
2600
|
-
file: saved[index].file,
|
|
2601
|
-
b64: Buffer.from(output.data).toString("base64"),
|
|
2602
|
-
mime: saved[index].mime,
|
|
2603
|
-
bytes: saved[index].bytes,
|
|
2604
|
-
url: saved[index].url,
|
|
2623
|
+
const outputs = await generateAudio(picked.channel, request, exec.signal);
|
|
2624
|
+
const audio = [];
|
|
2625
|
+
const saved = [];
|
|
2626
|
+
for (const [index, output] of outputs.entries()) {
|
|
2627
|
+
const stored = await saveAudioFile(output.data, output.mime, `generated-${index + 1}`);
|
|
2628
|
+
saved.push({
|
|
2629
|
+
id: stored.id,
|
|
2630
|
+
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
2631
|
+
file: stored.file,
|
|
2632
|
+
mime: stored.mime,
|
|
2633
|
+
bytes: stored.bytes,
|
|
2605
2634
|
...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
|
|
2606
|
-
})
|
|
2607
|
-
|
|
2608
|
-
|
|
2609
|
-
|
|
2610
|
-
|
|
2611
|
-
|
|
2612
|
-
|
|
2613
|
-
|
|
2614
|
-
|
|
2615
|
-
|
|
2616
|
-
|
|
2617
|
-
id:
|
|
2618
|
-
|
|
2619
|
-
mime: item.mime,
|
|
2620
|
-
...item.voiceId === void 0 ? {} : { voiceId: item.voiceId }
|
|
2621
|
-
})),
|
|
2622
|
-
type: libraryTypeOf(request.mode, args.library_type),
|
|
2623
|
-
...typeof args.library_name === "string" && args.library_name.trim() !== "" ? { name: args.library_name.trim() } : {},
|
|
2624
|
-
...Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag) => typeof tag === "string" && tag.trim() !== "").map((tag) => tag.trim()) } : {},
|
|
2625
|
-
provenance: {
|
|
2635
|
+
});
|
|
2636
|
+
audio.push({
|
|
2637
|
+
id: stored.id,
|
|
2638
|
+
url: `/api/dsh-audiogen/audio/${encodeURIComponent(stored.file)}`,
|
|
2639
|
+
mime: stored.mime,
|
|
2640
|
+
bytes: stored.bytes,
|
|
2641
|
+
...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
|
|
2642
|
+
});
|
|
2643
|
+
}
|
|
2644
|
+
try {
|
|
2645
|
+
await appendHistory({
|
|
2646
|
+
id: randomUUID(),
|
|
2647
|
+
createdAt: Date.now(),
|
|
2626
2648
|
mode: request.mode,
|
|
2627
|
-
prompt: request.prompt,
|
|
2628
|
-
channel: picked.channel.name,
|
|
2629
|
-
channelId: picked.channel.id,
|
|
2630
|
-
apiUrl: picked.channel.apiUrl,
|
|
2631
2649
|
model: picked.alias,
|
|
2632
|
-
|
|
2650
|
+
prompt: request.prompt,
|
|
2633
2651
|
...request.voice === void 0 ? {} : { voice: request.voice },
|
|
2652
|
+
...request.speed === void 0 ? {} : { speed: request.speed },
|
|
2653
|
+
...request.duration === void 0 ? {} : { duration: request.duration },
|
|
2654
|
+
...request.format === void 0 ? {} : { format: request.format },
|
|
2655
|
+
audio: outputs.map((output, index) => ({
|
|
2656
|
+
id: saved[index].id,
|
|
2657
|
+
file: saved[index].file,
|
|
2658
|
+
b64: Buffer.from(output.data).toString("base64"),
|
|
2659
|
+
mime: saved[index].mime,
|
|
2660
|
+
bytes: saved[index].bytes,
|
|
2661
|
+
url: saved[index].url,
|
|
2662
|
+
...output.voiceId === void 0 ? {} : { voiceId: output.voiceId }
|
|
2663
|
+
})),
|
|
2664
|
+
channelId: picked.channel.id,
|
|
2665
|
+
channel: picked.channel.name,
|
|
2634
2666
|
params: { ...request }
|
|
2635
|
-
}
|
|
2636
|
-
}
|
|
2637
|
-
|
|
2667
|
+
});
|
|
2668
|
+
} catch {}
|
|
2669
|
+
const wantSave = args.save_to_library === true || config.autoSaveToLibrary && args.save_to_library !== false;
|
|
2670
|
+
let resources;
|
|
2671
|
+
if (wantSave) try {
|
|
2672
|
+
resources = [(await saveToLibrary({
|
|
2673
|
+
audioFiles: saved.map((item) => ({
|
|
2674
|
+
id: item.id,
|
|
2675
|
+
file: item.file,
|
|
2676
|
+
mime: item.mime,
|
|
2677
|
+
...item.voiceId === void 0 ? {} : { voiceId: item.voiceId }
|
|
2678
|
+
})),
|
|
2679
|
+
type: libraryTypeOf(request.mode, args.library_type),
|
|
2680
|
+
...typeof args.library_name === "string" && args.library_name.trim() !== "" ? { name: args.library_name.trim() } : {},
|
|
2681
|
+
...Array.isArray(args.library_tags) ? { tags: args.library_tags.filter((tag) => typeof tag === "string" && tag.trim() !== "").map((tag) => tag.trim()) } : {},
|
|
2682
|
+
provenance: {
|
|
2683
|
+
mode: request.mode,
|
|
2684
|
+
prompt: request.prompt,
|
|
2685
|
+
channel: picked.channel.name,
|
|
2686
|
+
channelId: picked.channel.id,
|
|
2687
|
+
apiUrl: picked.channel.apiUrl,
|
|
2688
|
+
model: picked.alias,
|
|
2689
|
+
upstream: picked.upstream,
|
|
2690
|
+
...request.voice === void 0 ? {} : { voice: request.voice },
|
|
2691
|
+
params: { ...request }
|
|
2692
|
+
}
|
|
2693
|
+
})).id];
|
|
2694
|
+
} catch {}
|
|
2695
|
+
return {
|
|
2696
|
+
model: picked.alias,
|
|
2697
|
+
audio,
|
|
2698
|
+
...resources === void 0 ? {} : { resources }
|
|
2699
|
+
};
|
|
2700
|
+
} catch (error) {
|
|
2701
|
+
if (exec.signal?.aborted === true) throw error;
|
|
2702
|
+
return {
|
|
2703
|
+
model: picked.alias,
|
|
2704
|
+
audio: [],
|
|
2705
|
+
error: error instanceof Error ? error.message : String(error)
|
|
2706
|
+
};
|
|
2707
|
+
}
|
|
2708
|
+
};
|
|
2709
|
+
const requestedModels = Array.isArray(args.models) ? [...new Set(args.models.filter((item) => typeof item === "string" && item.trim() !== "").map((item) => item.trim()))] : [];
|
|
2710
|
+
if (requestedModels.length > 0 && mode !== "voice_design") {
|
|
2711
|
+
const groups = [];
|
|
2712
|
+
let succeeded = 0;
|
|
2713
|
+
for (const alias of requestedModels) {
|
|
2714
|
+
let picked;
|
|
2715
|
+
try {
|
|
2716
|
+
picked = resolveModel(config, alias);
|
|
2717
|
+
} catch (error) {
|
|
2718
|
+
groups.push({
|
|
2719
|
+
model: alias,
|
|
2720
|
+
audio: [],
|
|
2721
|
+
error: error instanceof Error ? error.message : String(error)
|
|
2722
|
+
});
|
|
2723
|
+
continue;
|
|
2724
|
+
}
|
|
2725
|
+
const group = await runOne(picked);
|
|
2726
|
+
groups.push(group);
|
|
2727
|
+
if (group.error === void 0) succeeded++;
|
|
2728
|
+
}
|
|
2638
2729
|
return {
|
|
2639
|
-
status: "completed",
|
|
2640
|
-
message:
|
|
2641
|
-
mode
|
|
2642
|
-
model:
|
|
2643
|
-
audio,
|
|
2644
|
-
|
|
2730
|
+
status: succeeded > 0 ? "completed" : "failed",
|
|
2731
|
+
message: succeeded > 0 ? `Generated ${succeeded}/${groups.length} model(s) with the same prompt for comparison. The audio files can be played/downloaded from the returned URLs.` : "All model generations failed.",
|
|
2732
|
+
mode,
|
|
2733
|
+
model: groups[0]?.model ?? requestedModels[0],
|
|
2734
|
+
audio: groups.flatMap((group) => group.audio),
|
|
2735
|
+
groups,
|
|
2736
|
+
...succeeded === 0 ? { error: groups.map((group) => `${group.model}: ${group.error ?? ""}`).filter((item) => !item.endsWith(": ")).join(";") } : {}
|
|
2645
2737
|
};
|
|
2646
|
-
}
|
|
2647
|
-
|
|
2738
|
+
}
|
|
2739
|
+
const one = await runOne(mode === "voice_design" ? (() => {
|
|
2740
|
+
const usable = config.channels.filter((channel) => channel.apiUrl.trim() !== "" && channel.apiKey.trim() !== "");
|
|
2741
|
+
const target = usable.find((channel) => channel.id === config.defaultChannelId) ?? usable[0];
|
|
2742
|
+
if (target === void 0) throw new AudioGenError("No usable audio channel is configured for voice design.", "no-channel-available");
|
|
2648
2743
|
return {
|
|
2649
|
-
|
|
2650
|
-
|
|
2651
|
-
|
|
2652
|
-
model: picked.alias,
|
|
2653
|
-
audio: [],
|
|
2654
|
-
error: error instanceof Error ? error.message : String(error)
|
|
2744
|
+
channel: target,
|
|
2745
|
+
alias: "",
|
|
2746
|
+
upstream: ""
|
|
2655
2747
|
};
|
|
2656
|
-
}
|
|
2748
|
+
})() : resolveModel(config, args.model));
|
|
2749
|
+
if (one.error !== void 0) return {
|
|
2750
|
+
status: "failed",
|
|
2751
|
+
message: "Audio generation failed.",
|
|
2752
|
+
mode,
|
|
2753
|
+
model: one.model,
|
|
2754
|
+
audio: [],
|
|
2755
|
+
error: one.error
|
|
2756
|
+
};
|
|
2757
|
+
return {
|
|
2758
|
+
status: "completed",
|
|
2759
|
+
message: "Audio generation completed. The audio files can be played/downloaded from the returned URLs.",
|
|
2760
|
+
mode,
|
|
2761
|
+
model: one.model,
|
|
2762
|
+
audio: one.audio,
|
|
2763
|
+
...one.resources === void 0 ? {} : { resources: one.resources }
|
|
2764
|
+
};
|
|
2657
2765
|
}
|
|
2658
2766
|
}));
|
|
2659
2767
|
const searchDisposer = ctx.tools.register(defineTool({
|
|
@@ -2831,6 +2939,29 @@ function guidanceFor(channels, defaultChannelId) {
|
|
|
2831
2939
|
}).join(";");
|
|
2832
2940
|
return `${AUDIOGEN_GUIDANCE} 当前渠道与模型:${table}。`;
|
|
2833
2941
|
}
|
|
2942
|
+
/**
|
|
2943
|
+
* 把随包分发的技能(skills/<id>/SKILL.md,含 frontmatter)同步到 DSH 用户技能根
|
|
2944
|
+
* `~/.dsh/skills/<id>/SKILL.md` —— DSH web 会话的 skill-filesystem(standard 等
|
|
2945
|
+
* preset 行)会扫描用户根,使会话可直接触发这些技能。仅创建缺失文件,绝不覆盖
|
|
2946
|
+
* 用户已有内容;任何失败仅告警,不影响插件本身。
|
|
2947
|
+
*/
|
|
2948
|
+
function syncBundledSkills() {
|
|
2949
|
+
try {
|
|
2950
|
+
const sourceRoot = join(dirname(dirname(fileURLToPath(import.meta.url))), "skills");
|
|
2951
|
+
if (existsSync(sourceRoot) !== true) return;
|
|
2952
|
+
const targetRoot = join(process.env.DSH_HOME ?? join(process.env.HOME ?? "", ".dsh"), "skills");
|
|
2953
|
+
for (const entry of readdirSync(sourceRoot, { withFileTypes: true })) {
|
|
2954
|
+
if (entry.isDirectory() !== true) continue;
|
|
2955
|
+
const sourceFile = join(sourceRoot, entry.name, "SKILL.md");
|
|
2956
|
+
if (existsSync(sourceFile) !== true) continue;
|
|
2957
|
+
const targetDir = join(targetRoot, entry.name);
|
|
2958
|
+
const targetFile = join(targetDir, "SKILL.md");
|
|
2959
|
+
if (existsSync(targetFile)) continue;
|
|
2960
|
+
mkdirSync(targetDir, { recursive: true });
|
|
2961
|
+
copyFileSync(sourceFile, targetFile);
|
|
2962
|
+
}
|
|
2963
|
+
} catch {}
|
|
2964
|
+
}
|
|
2834
2965
|
function normalizeChannels(value) {
|
|
2835
2966
|
if (!Array.isArray(value)) return [];
|
|
2836
2967
|
const out = [];
|
|
@@ -2862,6 +2993,7 @@ function normalizeChannels(value) {
|
|
|
2862
2993
|
return out;
|
|
2863
2994
|
}
|
|
2864
2995
|
function apply(ctx, config) {
|
|
2996
|
+
syncBundledSkills();
|
|
2865
2997
|
let current = () => config ?? {};
|
|
2866
2998
|
const resolve = () => {
|
|
2867
2999
|
const value = current() ?? {};
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-audiogen",
|
|
3
3
|
"description": "AI audio generation plugin for the dsh web GUI: multi-vendor TTS/music/sound-effect channels (OpenAI-compatible, ElevenLabs, MiniMax, Stability AI and custom), per-channel model/voice catalogs, Agent tool and a sidebar AI 音频 panel.",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.1",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "lib/index.js",
|
|
7
7
|
"exports": {
|
package/skills/design/SKILL.md
CHANGED
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: dsh-audiogen-voice-design
|
|
3
|
+
description: DSH AI 音频插件(dsh-audiogen)的音色设计技能:调用 generate_audio(mode=voice_design) 并指定厂商/渠道——MiniMax(POST /v1/voice_design,prompt + preview_text)或 ElevenLabs(POST /v1/text-to-voice/design,voice_description + 试听文本 100-1000 字符,过短自动生成;返回 previews[].audio_base_64 与 generated_voice_id 供后续 TTS 复用)。
|
|
4
|
+
whenToUse: 用户请求设计/创建新音色、音色试听,或触发 /audio:design 时使用。
|
|
5
|
+
---
|
|
1
6
|
# 音色/音效设计
|
|
2
7
|
|
|
3
8
|
## 触发
|
package/skills/music/SKILL.md
CHANGED
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: dsh-audiogen-music
|
|
3
|
+
description: DSH AI 音频插件(dsh-audiogen)的音乐生成技能:调用 generate_audio(mode=music);覆盖 MiniMax(music-3.0/2.6/cover,lyrics 歌词、is_instrumental 纯音乐、audio_setting 采样率 16000-44100/码率 32000-256000/格式 mp3-wav-pcm、时长)、ElevenLabs(/v1/music,music_v2,时长 3-600s、lyrics_text、force_instrumental)与 Stability(stable-audio 2/2.5/3,官方 v2beta 或 OpenAI 兼容 /v1/audio/speech 双通道,seed/steps/cfg_scale/duration)。
|
|
4
|
+
whenToUse: 用户请求生成音乐、配乐、BGM、纯音乐、歌曲,或触发 /audio:music 时使用;MiniMax 未给歌词且未要求纯音乐时,先补一段歌词或设置 is_instrumental。
|
|
5
|
+
---
|
|
1
6
|
# 音乐生成
|
|
2
7
|
|
|
3
8
|
## 触发
|
package/skills/sfx/SKILL.md
CHANGED
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: dsh-audiogen-sfx
|
|
3
|
+
description: DSH AI 音频插件(dsh-audiogen)的音效生成技能:调用 generate_audio(mode=sfx);覆盖 ElevenLabs(/v1/sound-generation,eleven_text_to_sound_v2,loop 无缝循环、prompt_influence 0-1、duration_seconds 0.5-30)与 MiniMax、Stability 等渠道的对应字段与常见错误处理。
|
|
4
|
+
whenToUse: 用户请求生成音效、提示音、环境音、UI 音,或触发 /audio:sfx 时使用。
|
|
5
|
+
---
|
|
1
6
|
# 音效生成
|
|
2
7
|
|
|
3
8
|
## 触发
|
package/skills/tts/SKILL.md
CHANGED
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: dsh-audiogen-tts
|
|
3
|
+
description: DSH AI 音频插件(dsh-audiogen)的 TTS 文本转语音技能:先确认渠道/模型/音色,再调用 generate_audio(mode=tts);包含 MiniMax 官方 t2a_v2 全字段(语速/音量/音调/情绪/采样率/码率/声道/发音词典/字幕/变声/双音色混合)、ElevenLabs 与 Stable Audio 的对应参数说明,以及常见错误(voice-required、网关 404 Invalid URL 等)的处理。
|
|
4
|
+
whenToUse: 用户提出朗读、配音、语音合成、TTS,或触发 /audio:tts 时使用;MiniMax 必须提供音色 voice_id。
|
|
5
|
+
---
|
|
1
6
|
# TTS 文本转语音
|
|
2
7
|
|
|
3
8
|
## 触发
|