@openclaw/fish-audio-speech 2026.9.1 → 2026.9.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/capability-catalog.js +5 -0
- package/dist/speech-provider.js +28 -21
- package/dist/tts.js +12 -10
- package/openclaw.plugin.json +11 -3
- package/package.json +4 -4
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import { buildFishAudioSpeechProvider } from "./speech-provider.js";
|
|
2
|
+
//#region extensions/fish-audio-speech/capability-catalog.ts
|
|
3
|
+
var capability_catalog_default = { speechProviders: [buildFishAudioSpeechProvider()] };
|
|
4
|
+
//#endregion
|
|
5
|
+
export { capability_catalog_default as default };
|
package/dist/speech-provider.js
CHANGED
|
@@ -1,8 +1,7 @@
|
|
|
1
1
|
import { FISH_AUDIO_STREAM_MAX_BYTES, fishAudioTts, fishAudioTtsStream, listFishAudioVoices, normalizeFishAudioBaseUrl } from "./tts.js";
|
|
2
|
-
import { resolveGeneratedMediaMaxBytes } from "openclaw/plugin-sdk/media-generation-runtime";
|
|
3
2
|
import { normalizeResolvedSecretInputString } from "openclaw/plugin-sdk/secret-input";
|
|
4
|
-
import {
|
|
5
|
-
import { asFiniteNumberInRange, asOptionalRecord, parseBooleanValue } from "openclaw/plugin-sdk/string-coerce-runtime";
|
|
3
|
+
import { parseSpeechDirectiveNumberOverride, resolveSpeechProviderApiKey } from "openclaw/plugin-sdk/speech-provider";
|
|
4
|
+
import { asBoolean, asFiniteNumberInRange, asOptionalRecord, normalizeOptionalString, parseBooleanValue } from "openclaw/plugin-sdk/string-coerce-runtime";
|
|
6
5
|
//#region extensions/fish-audio-speech/speech-provider.ts
|
|
7
6
|
const FISH_AUDIO_MODELS = [
|
|
8
7
|
"s2.1-pro-free",
|
|
@@ -14,13 +13,13 @@ const DEFAULT_MODEL = "s2.1-pro";
|
|
|
14
13
|
const DEFAULT_LATENCY = "balanced";
|
|
15
14
|
const DEFAULT_TIMEOUT_MS = 24e4;
|
|
16
15
|
function normalizeModel(value) {
|
|
17
|
-
const model =
|
|
16
|
+
const model = normalizeOptionalString(value);
|
|
18
17
|
if (!model) return DEFAULT_MODEL;
|
|
19
18
|
if (FISH_AUDIO_MODELS.some((candidate) => candidate === model)) return model;
|
|
20
19
|
throw new Error(`invalid Fish Audio model "${model}"`);
|
|
21
20
|
}
|
|
22
21
|
function normalizeLatency(value) {
|
|
23
|
-
const latency =
|
|
22
|
+
const latency = normalizeOptionalString(value)?.toLowerCase();
|
|
24
23
|
if (!latency) return DEFAULT_LATENCY;
|
|
25
24
|
if (latency === "low" || latency === "balanced" || latency === "normal") return latency;
|
|
26
25
|
throw new Error(`invalid Fish Audio latency "${latency}"`);
|
|
@@ -32,7 +31,7 @@ function normalizeNumber(value, min, max) {
|
|
|
32
31
|
});
|
|
33
32
|
}
|
|
34
33
|
function resolveReferenceId(raw) {
|
|
35
|
-
return
|
|
34
|
+
return normalizeOptionalString(raw?.speakerVoiceId ?? raw?.voiceId ?? raw?.referenceId);
|
|
36
35
|
}
|
|
37
36
|
function normalizeProviderConfig(rawConfig) {
|
|
38
37
|
const providers = asOptionalRecord(rawConfig.providers);
|
|
@@ -42,7 +41,7 @@ function normalizeProviderConfig(rawConfig) {
|
|
|
42
41
|
value: raw?.apiKey,
|
|
43
42
|
path: "tts.providers.fish-audio.apiKey"
|
|
44
43
|
}),
|
|
45
|
-
baseUrl: normalizeFishAudioBaseUrl(
|
|
44
|
+
baseUrl: normalizeFishAudioBaseUrl(normalizeOptionalString(raw?.baseUrl)),
|
|
46
45
|
model: normalizeModel(raw?.model ?? raw?.modelId),
|
|
47
46
|
referenceId: resolveReferenceId(raw),
|
|
48
47
|
latency: normalizeLatency(raw?.latency),
|
|
@@ -56,8 +55,8 @@ function readProviderConfig(config) {
|
|
|
56
55
|
const defaults = normalizeProviderConfig({});
|
|
57
56
|
const raw = asOptionalRecord(config) ?? {};
|
|
58
57
|
return {
|
|
59
|
-
apiKey:
|
|
60
|
-
baseUrl: normalizeFishAudioBaseUrl(
|
|
58
|
+
apiKey: normalizeOptionalString(raw.apiKey) ?? defaults.apiKey,
|
|
59
|
+
baseUrl: normalizeFishAudioBaseUrl(normalizeOptionalString(raw.baseUrl) ?? defaults.baseUrl),
|
|
61
60
|
model: normalizeModel(raw.model ?? raw.modelId ?? defaults.model),
|
|
62
61
|
referenceId: resolveReferenceId(raw) ?? defaults.referenceId,
|
|
63
62
|
latency: normalizeLatency(raw.latency ?? defaults.latency),
|
|
@@ -70,9 +69,9 @@ function readProviderConfig(config) {
|
|
|
70
69
|
function readOverrides(overrides) {
|
|
71
70
|
const raw = asOptionalRecord(overrides) ?? {};
|
|
72
71
|
return {
|
|
73
|
-
model:
|
|
72
|
+
model: normalizeOptionalString(raw.model ?? raw.modelId) ? normalizeModel(raw.model ?? raw.modelId) : void 0,
|
|
74
73
|
referenceId: resolveReferenceId(raw),
|
|
75
|
-
latency:
|
|
74
|
+
latency: normalizeOptionalString(raw.latency) ? normalizeLatency(raw.latency) : void 0,
|
|
76
75
|
speed: normalizeNumber(raw.speed, .5, 2),
|
|
77
76
|
temperature: normalizeNumber(raw.temperature, 0, 1),
|
|
78
77
|
topP: normalizeNumber(raw.topP ?? raw.top_p, 0, 1),
|
|
@@ -222,7 +221,6 @@ function resolveSynthesisRequest(req) {
|
|
|
222
221
|
topP: overrides.topP ?? config.topP,
|
|
223
222
|
normalize: overrides.normalize ?? config.normalize,
|
|
224
223
|
timeoutMs: req.timeoutMs,
|
|
225
|
-
maxBytes: resolveGeneratedMediaMaxBytes(req.cfg, "audio"),
|
|
226
224
|
...output
|
|
227
225
|
};
|
|
228
226
|
}
|
|
@@ -243,33 +241,37 @@ function buildFishAudioSpeechProvider() {
|
|
|
243
241
|
value: talkProviderConfig.apiKey,
|
|
244
242
|
path: "talk.providers.fish-audio.apiKey"
|
|
245
243
|
}) },
|
|
246
|
-
...
|
|
247
|
-
...
|
|
244
|
+
...normalizeOptionalString(talkProviderConfig.baseUrl) == null ? {} : { baseUrl: normalizeFishAudioBaseUrl(normalizeOptionalString(talkProviderConfig.baseUrl)) },
|
|
245
|
+
...normalizeOptionalString(talkProviderConfig.modelId ?? talkProviderConfig.model) == null ? {} : { model: normalizeModel(talkProviderConfig.modelId ?? talkProviderConfig.model) },
|
|
248
246
|
...resolveReferenceId(talkProviderConfig) == null ? {} : { referenceId: resolveReferenceId(talkProviderConfig) },
|
|
249
|
-
...
|
|
247
|
+
...normalizeOptionalString(talkProviderConfig.latency) == null ? {} : { latency: normalizeLatency(talkProviderConfig.latency) },
|
|
250
248
|
...normalizeNumber(talkProviderConfig.speed, .5, 2) == null ? {} : { speed: normalizeNumber(talkProviderConfig.speed, .5, 2) }
|
|
251
249
|
};
|
|
252
250
|
},
|
|
253
251
|
resolveTalkOverrides: ({ params }) => ({
|
|
254
|
-
...
|
|
252
|
+
...normalizeOptionalString(params.modelId ?? params.model) == null ? {} : { model: normalizeModel(params.modelId ?? params.model) },
|
|
255
253
|
...resolveReferenceId(params) == null ? {} : { referenceId: resolveReferenceId(params) },
|
|
256
254
|
...normalizeNumber(params.speed, .5, 2) == null ? {} : { speed: normalizeNumber(params.speed, .5, 2) }
|
|
257
255
|
}),
|
|
258
256
|
listVoices: async (req) => {
|
|
259
257
|
const config = readProviderConfig(req.providerConfig ?? {});
|
|
260
|
-
const apiKey = resolveApiKey(
|
|
258
|
+
const apiKey = resolveApiKey(normalizeOptionalString(req.apiKey) ?? config.apiKey);
|
|
261
259
|
if (!apiKey) throw new Error("Fish Audio API key missing");
|
|
262
260
|
return await listFishAudioVoices({
|
|
263
261
|
apiKey,
|
|
264
|
-
baseUrl: normalizeFishAudioBaseUrl(
|
|
262
|
+
baseUrl: normalizeFishAudioBaseUrl(normalizeOptionalString(req.baseUrl) ?? config.baseUrl),
|
|
265
263
|
timeoutMs: req.timeoutMs ?? DEFAULT_TIMEOUT_MS
|
|
266
264
|
});
|
|
267
265
|
},
|
|
268
266
|
isConfigured: ({ providerConfig }) => Boolean(resolveApiKey(readProviderConfig(providerConfig).apiKey)),
|
|
269
267
|
synthesize: async (req) => {
|
|
270
268
|
const params = resolveSynthesisRequest(req);
|
|
269
|
+
const { resolveGeneratedMediaMaxBytes } = await import("openclaw/plugin-sdk/media-generation-runtime");
|
|
271
270
|
return {
|
|
272
|
-
audioBuffer: await fishAudioTts(
|
|
271
|
+
audioBuffer: await fishAudioTts({
|
|
272
|
+
...params,
|
|
273
|
+
maxBytes: resolveGeneratedMediaMaxBytes(req.cfg, "audio")
|
|
274
|
+
}),
|
|
273
275
|
outputFormat: params.format,
|
|
274
276
|
fileExtension: params.fileExtension,
|
|
275
277
|
voiceCompatible: params.voiceCompatible
|
|
@@ -277,9 +279,10 @@ function buildFishAudioSpeechProvider() {
|
|
|
277
279
|
},
|
|
278
280
|
streamSynthesize: async (req) => {
|
|
279
281
|
const params = resolveSynthesisRequest(req);
|
|
282
|
+
const { resolveGeneratedMediaMaxBytes } = await import("openclaw/plugin-sdk/media-generation-runtime");
|
|
280
283
|
const stream = await fishAudioTtsStream({
|
|
281
284
|
...params,
|
|
282
|
-
maxBytes: Math.min(
|
|
285
|
+
maxBytes: Math.min(resolveGeneratedMediaMaxBytes(req.cfg, "audio"), FISH_AUDIO_STREAM_MAX_BYTES)
|
|
283
286
|
});
|
|
284
287
|
return {
|
|
285
288
|
audioStream: stream.audioStream,
|
|
@@ -294,8 +297,12 @@ function buildFishAudioSpeechProvider() {
|
|
|
294
297
|
...req,
|
|
295
298
|
target: "telephony"
|
|
296
299
|
});
|
|
300
|
+
const { resolveGeneratedMediaMaxBytes } = await import("openclaw/plugin-sdk/media-generation-runtime");
|
|
297
301
|
return {
|
|
298
|
-
audioBuffer: await fishAudioTts(
|
|
302
|
+
audioBuffer: await fishAudioTts({
|
|
303
|
+
...params,
|
|
304
|
+
maxBytes: resolveGeneratedMediaMaxBytes(req.cfg, "audio")
|
|
305
|
+
}),
|
|
299
306
|
outputFormat: "pcm",
|
|
300
307
|
sampleRate: 8e3
|
|
301
308
|
};
|
package/dist/tts.js
CHANGED
|
@@ -1,9 +1,5 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import { createBoundedProviderBinaryStream } from "openclaw/plugin-sdk/provider-binary-stream";
|
|
4
|
-
import { assertOkOrThrowProviderError, assertProviderBinaryResponseContent, readProviderBinaryResponse, readProviderJsonResponse } from "openclaw/plugin-sdk/provider-http";
|
|
5
|
-
import { trimToUndefined } from "openclaw/plugin-sdk/speech";
|
|
6
|
-
import { fetchWithSsrFGuard, ssrfPolicyFromHttpBaseUrlAllowedHostname } from "openclaw/plugin-sdk/ssrf-runtime";
|
|
1
|
+
import { MAX_AUDIO_BYTES } from "openclaw/plugin-sdk/speech-provider";
|
|
2
|
+
import { asOptionalRecord, normalizeOptionalString } from "openclaw/plugin-sdk/string-coerce-runtime";
|
|
7
3
|
//#region extensions/fish-audio-speech/tts.ts
|
|
8
4
|
const FISH_AUDIO_BASE_URL = "https://api.fish.audio";
|
|
9
5
|
const FISH_AUDIO_VOICES_MAX_BYTES = 2097152;
|
|
@@ -28,6 +24,7 @@ function buildFishAudioRequestBody(params) {
|
|
|
28
24
|
}
|
|
29
25
|
async function requestFishAudioTts(params) {
|
|
30
26
|
const baseUrl = normalizeFishAudioBaseUrl(params.baseUrl);
|
|
27
|
+
const { fetchWithSsrFGuard, ssrfPolicyFromHttpBaseUrlAllowedHostname } = await import("openclaw/plugin-sdk/ssrf-runtime");
|
|
31
28
|
return await fetchWithSsrFGuard({
|
|
32
29
|
url: `${baseUrl}/v1/tts`,
|
|
33
30
|
init: {
|
|
@@ -45,6 +42,7 @@ async function requestFishAudioTts(params) {
|
|
|
45
42
|
});
|
|
46
43
|
}
|
|
47
44
|
async function fishAudioTts(params) {
|
|
45
|
+
const { assertOkOrThrowProviderError, readProviderBinaryResponse } = await import("openclaw/plugin-sdk/provider-http");
|
|
48
46
|
const { response, release } = await requestFishAudioTts(params);
|
|
49
47
|
try {
|
|
50
48
|
await assertOkOrThrowProviderError(response, "Fish Audio TTS API error");
|
|
@@ -54,6 +52,8 @@ async function fishAudioTts(params) {
|
|
|
54
52
|
}
|
|
55
53
|
}
|
|
56
54
|
async function fishAudioTtsStream(params) {
|
|
55
|
+
const { createBoundedProviderBinaryStream } = await import("openclaw/plugin-sdk/provider-binary-stream");
|
|
56
|
+
const { assertOkOrThrowProviderError, assertProviderBinaryResponseContent } = await import("openclaw/plugin-sdk/provider-http");
|
|
57
57
|
const { response, release } = await requestFishAudioTts(params);
|
|
58
58
|
let handedOff = false;
|
|
59
59
|
try {
|
|
@@ -77,15 +77,15 @@ async function fishAudioTtsStream(params) {
|
|
|
77
77
|
}
|
|
78
78
|
function parseVoiceItem(value) {
|
|
79
79
|
const item = asOptionalRecord(value);
|
|
80
|
-
const id =
|
|
80
|
+
const id = normalizeOptionalString(item?.["_id"]);
|
|
81
81
|
if (!id) return;
|
|
82
82
|
const languages = Array.isArray(item?.languages) ? item.languages.flatMap((entry) => typeof entry === "string" && entry.trim() ? [entry.trim()] : []) : [];
|
|
83
83
|
const tags = Array.isArray(item?.tags) ? item.tags.flatMap((entry) => typeof entry === "string" && entry.trim() ? [entry.trim()] : []) : [];
|
|
84
84
|
return {
|
|
85
85
|
id,
|
|
86
|
-
name:
|
|
87
|
-
description:
|
|
88
|
-
category:
|
|
86
|
+
name: normalizeOptionalString(item?.title),
|
|
87
|
+
description: normalizeOptionalString(item?.description),
|
|
88
|
+
category: normalizeOptionalString(item?.visibility),
|
|
89
89
|
locale: languages[0],
|
|
90
90
|
personalities: tags.length > 0 ? tags : void 0
|
|
91
91
|
};
|
|
@@ -97,6 +97,8 @@ async function requestVoicePage(params) {
|
|
|
97
97
|
url.searchParams.set("page_number", String(params.pageNumber));
|
|
98
98
|
if (params.self) url.searchParams.set("self", "true");
|
|
99
99
|
else url.searchParams.set("sort_by", "score");
|
|
100
|
+
const { assertOkOrThrowProviderError, readProviderJsonResponse } = await import("openclaw/plugin-sdk/provider-http");
|
|
101
|
+
const { fetchWithSsrFGuard, ssrfPolicyFromHttpBaseUrlAllowedHostname } = await import("openclaw/plugin-sdk/ssrf-runtime");
|
|
100
102
|
const { response, release } = await fetchWithSsrFGuard({
|
|
101
103
|
url: url.toString(),
|
|
102
104
|
init: { headers: { Authorization: `Bearer ${params.apiKey}` } },
|
package/openclaw.plugin.json
CHANGED
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
{
|
|
2
2
|
"id": "fish-audio-speech",
|
|
3
|
-
"
|
|
3
|
+
"capabilityCatalogEntry": "./dist/capability-catalog.js",
|
|
4
|
+
"legacyPluginIds": [
|
|
5
|
+
"fish-audio"
|
|
6
|
+
],
|
|
4
7
|
"activation": {
|
|
5
8
|
"onStartup": false
|
|
6
9
|
},
|
|
@@ -10,12 +13,17 @@
|
|
|
10
13
|
"providers": [
|
|
11
14
|
{
|
|
12
15
|
"id": "fish-audio",
|
|
13
|
-
"envVars": [
|
|
16
|
+
"envVars": [
|
|
17
|
+
"FISH_API_KEY",
|
|
18
|
+
"FISH_AUDIO_API_KEY"
|
|
19
|
+
]
|
|
14
20
|
}
|
|
15
21
|
]
|
|
16
22
|
},
|
|
17
23
|
"contracts": {
|
|
18
|
-
"speechProviders": [
|
|
24
|
+
"speechProviders": [
|
|
25
|
+
"fish-audio"
|
|
26
|
+
]
|
|
19
27
|
},
|
|
20
28
|
"configSchema": {
|
|
21
29
|
"type": "object",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@openclaw/fish-audio-speech",
|
|
3
|
-
"version": "2026.9.
|
|
3
|
+
"version": "2026.9.2",
|
|
4
4
|
"description": "OpenClaw Fish Audio speech plugin.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -18,10 +18,10 @@
|
|
|
18
18
|
"minHostVersion": ">=2026.7.2"
|
|
19
19
|
},
|
|
20
20
|
"compat": {
|
|
21
|
-
"pluginApi": ">=2026.9.
|
|
21
|
+
"pluginApi": ">=2026.9.2"
|
|
22
22
|
},
|
|
23
23
|
"build": {
|
|
24
|
-
"openclawVersion": "2026.9.
|
|
24
|
+
"openclawVersion": "2026.9.2",
|
|
25
25
|
"bundledDist": false
|
|
26
26
|
},
|
|
27
27
|
"release": {
|
|
@@ -38,7 +38,7 @@
|
|
|
38
38
|
"README.md"
|
|
39
39
|
],
|
|
40
40
|
"peerDependencies": {
|
|
41
|
-
"openclaw": ">=2026.9.
|
|
41
|
+
"openclaw": ">=2026.9.2"
|
|
42
42
|
},
|
|
43
43
|
"peerDependenciesMeta": {
|
|
44
44
|
"openclaw": {
|