@tanstack/ai-byteplus 0.3.7 → 0.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/tts.d.ts +17 -1
- package/dist/esm/adapters/tts.js +79 -6
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/index.js +2 -2
- package/package.json +5 -5
- package/src/adapters/tts.ts +130 -11
- package/src/index.ts +1 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { BaseTTSAdapter } from '@tanstack/ai/adapters';
|
|
2
|
-
import { TTSOptions } from '@tanstack/ai';
|
|
2
|
+
import { TTSCapabilities, TTSOptions, TTSTurn } from '@tanstack/ai';
|
|
3
3
|
import { InternalLogger } from '@tanstack/ai/adapter-internals';
|
|
4
4
|
import { BytePlusVoiceConfig } from '../utils/client.js';
|
|
5
5
|
import { BytePlusTTSModel } from '../model-meta.js';
|
|
@@ -19,6 +19,12 @@ export declare const BYTEPLUS_DEFAULT_TTS_SPEAKER = "en_female_stokie_uranus_big
|
|
|
19
19
|
* when `speech_rate` slows it down.
|
|
20
20
|
*/
|
|
21
21
|
export declare const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
|
|
22
|
+
/**
|
|
23
|
+
* Hard cap on the `references` array, and therefore on the number of distinct
|
|
24
|
+
* voices one dialogue request can use. The markers that cite them in
|
|
25
|
+
* `text_prompt` run `@Audio1`..`@Audio3`.
|
|
26
|
+
*/
|
|
27
|
+
export declare const BYTEPLUS_TTS_MAX_REFERENCES = 3;
|
|
22
28
|
/**
|
|
23
29
|
* BytePlus Seed Speech text-to-speech adapter.
|
|
24
30
|
*
|
|
@@ -52,6 +58,12 @@ export declare const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
|
|
|
52
58
|
*/
|
|
53
59
|
export declare class BytePlusTTSAdapter<TModel extends BytePlusTTSModel = BytePlusTTSModel> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {
|
|
54
60
|
readonly name: "byteplus";
|
|
61
|
+
/**
|
|
62
|
+
* Seed Audio 1.0 does multi-role dialogue in one pass, addressed through
|
|
63
|
+
* `references` (capped at 3 entries), and returns word/sentence timings
|
|
64
|
+
* when `audio_config.enable_subtitle` is set.
|
|
65
|
+
*/
|
|
66
|
+
readonly capabilities: TTSCapabilities;
|
|
55
67
|
private readonly apiKey;
|
|
56
68
|
private readonly baseURL;
|
|
57
69
|
private readonly defaultHeaders;
|
|
@@ -70,10 +82,14 @@ export declare class BytePlusTTSAdapter<TModel extends BytePlusTTSModel = BytePl
|
|
|
70
82
|
export declare function buildTTSRequestBody(options: {
|
|
71
83
|
model: string;
|
|
72
84
|
text: string;
|
|
85
|
+
/** Dialogue turns, when the caller asked for multi-role synthesis. */
|
|
86
|
+
turns?: Array<TTSTurn>;
|
|
73
87
|
voice: string | undefined;
|
|
74
88
|
format: TTSOptions['format'] | undefined;
|
|
75
89
|
speed: number | undefined;
|
|
76
90
|
modelOptions: BytePlusTTSProviderOptions | undefined;
|
|
91
|
+
/** Core `timestamps: true` — turns on `audio_config.enable_subtitle`. */
|
|
92
|
+
timestamps?: boolean;
|
|
77
93
|
logger: InternalLogger;
|
|
78
94
|
}): {
|
|
79
95
|
body: BytePlusTTSCreateRequest;
|
package/dist/esm/adapters/tts.js
CHANGED
|
@@ -51,6 +51,12 @@ var BYTEPLUS_DEFAULT_TTS_SPEAKER = "en_female_stokie_uranus_bigtts";
|
|
|
51
51
|
*/
|
|
52
52
|
var BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
|
|
53
53
|
/**
|
|
54
|
+
* Hard cap on the `references` array, and therefore on the number of distinct
|
|
55
|
+
* voices one dialogue request can use. The markers that cite them in
|
|
56
|
+
* `text_prompt` run `@Audio1`..`@Audio3`.
|
|
57
|
+
*/
|
|
58
|
+
var BYTEPLUS_TTS_MAX_REFERENCES = 3;
|
|
59
|
+
/**
|
|
54
60
|
* BytePlus Seed Speech text-to-speech adapter.
|
|
55
61
|
*
|
|
56
62
|
* Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.
|
|
@@ -83,6 +89,15 @@ var BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
|
|
|
83
89
|
*/
|
|
84
90
|
var BytePlusTTSAdapter = class extends BaseTTSAdapter {
|
|
85
91
|
name = "byteplus";
|
|
92
|
+
/**
|
|
93
|
+
* Seed Audio 1.0 does multi-role dialogue in one pass, addressed through
|
|
94
|
+
* `references` (capped at 3 entries), and returns word/sentence timings
|
|
95
|
+
* when `audio_config.enable_subtitle` is set.
|
|
96
|
+
*/
|
|
97
|
+
capabilities = {
|
|
98
|
+
maxSpeakers: 3,
|
|
99
|
+
timestamps: true
|
|
100
|
+
};
|
|
86
101
|
apiKey;
|
|
87
102
|
baseURL;
|
|
88
103
|
defaultHeaders;
|
|
@@ -96,7 +111,7 @@ var BytePlusTTSAdapter = class extends BaseTTSAdapter {
|
|
|
96
111
|
this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis);
|
|
97
112
|
}
|
|
98
113
|
async generateSpeech(options) {
|
|
99
|
-
const { logger, model, text, voice, format, speed, modelOptions } = options;
|
|
114
|
+
const { logger, model, text, voice, format, speed, modelOptions, turns } = options;
|
|
100
115
|
logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {
|
|
101
116
|
provider: "byteplus",
|
|
102
117
|
model
|
|
@@ -104,10 +119,12 @@ var BytePlusTTSAdapter = class extends BaseTTSAdapter {
|
|
|
104
119
|
const { body, audioFormat, sampleRate } = buildTTSRequestBody({
|
|
105
120
|
model,
|
|
106
121
|
text,
|
|
122
|
+
turns,
|
|
107
123
|
voice,
|
|
108
124
|
format,
|
|
109
125
|
speed,
|
|
110
126
|
modelOptions,
|
|
127
|
+
timestamps: options.timestamps,
|
|
111
128
|
logger
|
|
112
129
|
});
|
|
113
130
|
try {
|
|
@@ -135,6 +152,7 @@ var BytePlusTTSAdapter = class extends BaseTTSAdapter {
|
|
|
135
152
|
...duration !== void 0 && { duration },
|
|
136
153
|
...originalDuration !== void 0 && { originalDuration },
|
|
137
154
|
...data.subtitle !== void 0 && { subtitle: data.subtitle },
|
|
155
|
+
...toAlignmentFields(data.subtitle),
|
|
138
156
|
...data.url !== void 0 && { url: data.url }
|
|
139
157
|
};
|
|
140
158
|
} catch (error) {
|
|
@@ -155,7 +173,7 @@ var BytePlusTTSAdapter = class extends BaseTTSAdapter {
|
|
|
155
173
|
* `contentType`.
|
|
156
174
|
*/
|
|
157
175
|
function buildTTSRequestBody(options) {
|
|
158
|
-
const { model, text, voice, format, speed, modelOptions, logger } = options;
|
|
176
|
+
const { model, text, turns, voice, format, speed, modelOptions, logger } = options;
|
|
159
177
|
const audioFormat = pickAudioFormat(modelOptions?.format, format, logger);
|
|
160
178
|
const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE;
|
|
161
179
|
const audioConfig = {
|
|
@@ -164,13 +182,15 @@ function buildTTSRequestBody(options) {
|
|
|
164
182
|
};
|
|
165
183
|
if (modelOptions?.pitch_rate !== void 0) audioConfig.pitch_rate = modelOptions.pitch_rate;
|
|
166
184
|
if (modelOptions?.loudness_rate !== void 0) audioConfig.loudness_rate = modelOptions.loudness_rate;
|
|
167
|
-
|
|
185
|
+
const enableSubtitle = modelOptions?.enable_subtitle ?? (options.timestamps || void 0);
|
|
186
|
+
if (enableSubtitle !== void 0) audioConfig.enable_subtitle = enableSubtitle;
|
|
168
187
|
const speechRate = modelOptions?.speech_rate ?? (speed !== void 0 ? toSpeechRate(speed, logger) : void 0);
|
|
169
188
|
if (speechRate !== void 0) audioConfig.speech_rate = speechRate;
|
|
189
|
+
const dialogue = turns ? buildDialoguePrompt(turns) : void 0;
|
|
170
190
|
const body = {
|
|
171
191
|
model,
|
|
172
|
-
[TTS_TEXT_FIELD]: text,
|
|
173
|
-
references: modelOptions?.references ?? [{ speaker: modelOptions?.speaker ?? voice ?? "en_female_stokie_uranus_bigtts" }],
|
|
192
|
+
[TTS_TEXT_FIELD]: dialogue?.textPrompt ?? text,
|
|
193
|
+
references: modelOptions?.references ?? dialogue?.references ?? [{ speaker: modelOptions?.speaker ?? voice ?? "en_female_stokie_uranus_bigtts" }],
|
|
174
194
|
audio_config: audioConfig
|
|
175
195
|
};
|
|
176
196
|
if (modelOptions?.watermark !== void 0) body.watermark = typeof modelOptions.watermark === "boolean" ? { aigc_watermark: modelOptions.watermark } : modelOptions.watermark;
|
|
@@ -256,6 +276,59 @@ function getContentType(format, sampleRate) {
|
|
|
256
276
|
}
|
|
257
277
|
}
|
|
258
278
|
/**
|
|
279
|
+
* Turn dialogue turns into the one-prompt form Seed Audio 1.0 takes: each
|
|
280
|
+
* distinct voice becomes a `references` entry, and every line of the script
|
|
281
|
+
* cites its voice by the reference's 1-based position (`@Audio1`..`@Audio3`),
|
|
282
|
+
* which is the marker convention the reference array already uses.
|
|
283
|
+
*
|
|
284
|
+
* **The prompt form is unprobed.** The flat `references[]` member shape is
|
|
285
|
+
* confirmed (see `BytePlusTTSReference`), and the model card documents
|
|
286
|
+
* multi-role dialogue and positional reference markers, but no worked
|
|
287
|
+
* multi-speaker example was available and no Seed Speech key existed to
|
|
288
|
+
* confirm how lines should cite their voice.
|
|
289
|
+
*/
|
|
290
|
+
function buildDialoguePrompt(turns) {
|
|
291
|
+
const voices = [...new Set(turns.map((turn) => turn.voice))];
|
|
292
|
+
return {
|
|
293
|
+
textPrompt: turns.map((turn) => `@Audio${voices.indexOf(turn.voice) + 1}: ${turn.text}`).join("\n"),
|
|
294
|
+
references: voices.map((speaker) => ({ speaker }))
|
|
295
|
+
};
|
|
296
|
+
}
|
|
297
|
+
/**
|
|
298
|
+
* Map the BytePlus `subtitle` block onto the cross-provider `alignment` /
|
|
299
|
+
* `segments` fields.
|
|
300
|
+
*
|
|
301
|
+
* `words` is the finest granularity Seed Speech reports, so it becomes
|
|
302
|
+
* `alignment`; `sentences` is utterance segmentation, so it becomes
|
|
303
|
+
* `segments`. **Subtitle times are milliseconds** while the rest of the
|
|
304
|
+
* response is seconds — the conversion happens here, once.
|
|
305
|
+
*
|
|
306
|
+
* The raw block stays on {@link BytePlusTTSResult.subtitle} for callers
|
|
307
|
+
* already reading it.
|
|
308
|
+
*/
|
|
309
|
+
function toAlignmentFields(subtitle) {
|
|
310
|
+
if (!subtitle) return {};
|
|
311
|
+
const fields = {};
|
|
312
|
+
const words = subtitle.words?.filter(isTimedEntry);
|
|
313
|
+
if (words && words.length > 0) fields.alignment = {
|
|
314
|
+
unit: "word",
|
|
315
|
+
texts: words.map((word) => word.text ?? ""),
|
|
316
|
+
startSeconds: words.map((word) => word.start_time / 1e3),
|
|
317
|
+
endSeconds: words.map((word) => word.end_time / 1e3)
|
|
318
|
+
};
|
|
319
|
+
const sentences = subtitle.sentences?.filter(isTimedEntry);
|
|
320
|
+
if (sentences && sentences.length > 0) fields.segments = sentences.map((sentence) => ({
|
|
321
|
+
startSeconds: sentence.start_time / 1e3,
|
|
322
|
+
endSeconds: sentence.end_time / 1e3,
|
|
323
|
+
...sentence.text !== void 0 && { text: sentence.text }
|
|
324
|
+
}));
|
|
325
|
+
return fields;
|
|
326
|
+
}
|
|
327
|
+
/** Both times present, so the entry can be placed on the timeline. */
|
|
328
|
+
function isTimedEntry(entry) {
|
|
329
|
+
return typeof entry.start_time === "number" && typeof entry.end_time === "number";
|
|
330
|
+
}
|
|
331
|
+
/**
|
|
259
332
|
* Coerce a `duration` / `original_duration` field to a usable number of
|
|
260
333
|
* seconds.
|
|
261
334
|
*
|
|
@@ -302,6 +375,6 @@ function byteplusSpeech(model, config) {
|
|
|
302
375
|
return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config);
|
|
303
376
|
}
|
|
304
377
|
//#endregion
|
|
305
|
-
export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BytePlusTTSAdapter, buildTTSRequestBody, byteplusSpeech, createBytePlusSpeech, getContentType, toDurationSeconds, toSpeechRate };
|
|
378
|
+
export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MAX_REFERENCES, BytePlusTTSAdapter, buildTTSRequestBody, byteplusSpeech, createBytePlusSpeech, getContentType, toDurationSeconds, toSpeechRate };
|
|
306
379
|
|
|
307
380
|
//# sourceMappingURL=tts.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"tts.js","names":[],"sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport { generateId } from '@tanstack/ai-utils'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport {\n BYTEPLUS_VOICE_BASE_URL,\n bytePlusVoiceError,\n bytePlusVoiceHeaders,\n getBytePlusVoiceApiKeyFromEnv,\n readJsonBody,\n withBytePlusVoiceDefaults,\n} from '../utils/client'\nimport type { TTSOptions } from '@tanstack/ai'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { BytePlusVoiceConfig } from '../utils/client'\nimport type { BytePlusTTSModel } from '../model-meta'\nimport type {\n BytePlusTTSAudioConfig,\n BytePlusTTSAudioFormat,\n BytePlusTTSCreateRequest,\n BytePlusTTSCreateResponse,\n} from '../audio/wire-types'\nimport type {\n BytePlusTTSProviderOptions,\n BytePlusTTSResult,\n} from '../audio/tts-provider-options'\n\n/** Path of the synchronous Seed Speech synthesis endpoint. */\nconst TTS_CREATE_PATH = '/api/v3/tts/create'\n\n/**\n * Name of the request field carrying the text to speak.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** The endpoint\n * schema (docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01) lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`; there is no request-side `text`. The `text` spelling belongs\n * to the *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits\n * under `req_params.text`. The name stays isolated here so the two spellings\n * never get conflated.\n */\nconst TTS_TEXT_FIELD = 'text_prompt' satisfies keyof BytePlusTTSCreateRequest\n\n/**\n * Sample rate used when the caller doesn't pick one. The endpoint documents a\n * default of 40000, which is not among the rates it accepts — so the adapter\n * never relies on the server default and always sends this instead.\n */\nconst DEFAULT_SAMPLE_RATE = 24000\n\n/**\n * True when a Seed Speech envelope's `code` means success.\n *\n * Seed Speech uses `0` for success and a flat numeric code otherwise\n * (`45000010 Invalid X-Api-Key`). The success envelope has not been confirmed\n * against a live key, so `code` is accepted as a number, its string form, or\n * absent — an envelope that omits `code` entirely is treated as success, which\n * is what the HTTP status already told us.\n */\nfunction isZeroCode(code: number | string | undefined): boolean {\n if (code === undefined) return true\n return Number(code) === 0\n}\n\n/**\n * Voice used when neither `TTSOptions.voice` nor `modelOptions.speaker` is\n * set — the English female \"Stokie\" voice from the TTS 2.0 generation.\n */\nexport const BYTEPLUS_DEFAULT_TTS_SPEAKER = 'en_female_stokie_uranus_bigtts'\n\n/**\n * Hard cap on a single Seed Speech synthesis, in seconds. Longer scripts must\n * be split across calls and stitched client-side.\n *\n * The cap applies to the *pre-rate* length the service bills on — the\n * `original_duration` it returns. The delivered clip can run longer than this\n * when `speech_rate` slows it down.\n */\nexport const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120\n\n/**\n * BytePlus Seed Speech text-to-speech adapter.\n *\n * Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.\n * Two things differ from the Ark-hosted adapters in this package:\n *\n * - **Separate API key.** Seed Speech authenticates with `X-Api-Key` and its\n * own key (`BYTEPLUS_VOICE_API_KEY`). An `ARK_API_KEY` sent here is\n * rejected with `45000010 Invalid X-Api-Key`.\n * - **120 s output cap.** A single call synthesises at most\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} seconds of audio; longer scripts\n * have to be split and stitched client-side. The cap is measured before\n * `speech_rate` is applied, so a slowed clip can play for longer than that\n * — `result.duration` is the delivered length and `result.originalDuration`\n * is the metered one.\n *\n * Streaming synthesis (`tts/unidirectional` and the WebSocket API) is not\n * covered by this adapter — note that endpoint spells the text field `text`,\n * while this one uses `text_prompt`.\n *\n * @example\n * ```ts\n * const adapter = byteplusSpeech('seed-audio-1.0')\n * const result = await generateSpeech({\n * adapter,\n * text: 'welcome to the guitar store',\n * voice: 'en_female_stokie_uranus_bigtts',\n * format: 'mp3',\n * })\n * ```\n */\nexport class BytePlusTTSAdapter<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {\n readonly name = 'byteplus' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n private readonly fetchImpl: typeof fetch\n\n constructor(model: TModel, config: BytePlusVoiceConfig) {\n super(model, config)\n const resolved = withBytePlusVoiceDefaults(config)\n this.apiKey = resolved.apiKey\n this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL\n this.defaultHeaders = resolved.defaultHeaders ?? {}\n this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)\n }\n\n async generateSpeech(\n options: TTSOptions<BytePlusTTSProviderOptions>,\n ): Promise<BytePlusTTSResult> {\n const { logger, model, text, voice, format, speed, modelOptions } = options\n\n logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {\n provider: 'byteplus',\n model,\n })\n\n const { body, audioFormat, sampleRate } = buildTTSRequestBody({\n model,\n text,\n voice,\n format,\n speed,\n modelOptions,\n logger,\n })\n\n try {\n const response = await this.fetchImpl(\n `${this.baseURL}${TTS_CREATE_PATH}`,\n {\n method: 'POST',\n headers: bytePlusVoiceHeaders(this.apiKey, {\n ...this.defaultHeaders,\n // Client-generated per-request id. BytePlus echoes it in their\n // request logs, which is what support asks for when diagnosing a\n // synthesis failure.\n 'X-Api-Request-Id': newRequestId(),\n }),\n body: JSON.stringify(body),\n },\n )\n\n const payload = await readJsonBody(response)\n\n if (!response.ok) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n const data = payload as BytePlusTTSCreateResponse\n\n // Seed Speech reports status in the body, not only in the HTTP status:\n // a 200 can carry a non-zero `code`. Check it before looking at `audio`,\n // because a failed call may still return a partial or placeholder\n // payload that would otherwise be handed back as if it were valid.\n //\n // `code` is accepted as a number *or* a string. The success envelope was\n // never confirmed against a live key (no voice key yet — see\n // `audio/wire-types.ts`), and `readStringField` already tolerates both\n // forms when rendering the error, so requiring a number here would let\n // `{\"code\": \"45000010\"}` through both this gate and the one below.\n if (!isZeroCode(data.code)) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n // Belt and braces for a 200 that reports success but carries nothing to\n // play. Say that the adapter rejected it, rather than reusing the\n // envelope-error phrasing — a bare \"failed (200)\" gives no hint that the\n // response was well-formed and simply empty.\n if (typeof data.audio !== 'string' || data.audio.length === 0) {\n throw new Error(\n `BytePlus Seed Speech text-to-speech returned a success response ` +\n `with no audio (model ${model}).`,\n )\n }\n\n const duration = toDurationSeconds(data.duration)\n const originalDuration = toDurationSeconds(data.original_duration)\n\n return {\n id: generateId(this.name),\n model,\n audio: data.audio,\n format: audioFormat,\n contentType: getContentType(audioFormat, sampleRate),\n ...(duration !== undefined && { duration }),\n ...(originalDuration !== undefined && { originalDuration }),\n ...(data.subtitle !== undefined && { subtitle: data.subtitle }),\n ...(data.url !== undefined && { url: data.url }),\n }\n } catch (error) {\n logger.errors('byteplus.generateSpeech fatal', {\n error: toRunErrorPayload(error, 'byteplus.generateSpeech failed'),\n source: 'byteplus.generateSpeech',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the JSON body for `POST /api/v3/tts/create`, resolving the voice,\n * output format and rate fields in one place.\n *\n * Returns the request `body`, the resolved `audioFormat` and the\n * `sampleRate`, which the caller reports on the result and turns into a\n * `contentType`.\n */\nexport function buildTTSRequestBody(options: {\n model: string\n text: string\n voice: string | undefined\n format: TTSOptions['format'] | undefined\n speed: number | undefined\n modelOptions: BytePlusTTSProviderOptions | undefined\n logger: InternalLogger\n}): {\n body: BytePlusTTSCreateRequest\n audioFormat: BytePlusTTSAudioFormat\n sampleRate: number\n} {\n const { model, text, voice, format, speed, modelOptions, logger } = options\n\n const audioFormat = pickAudioFormat(modelOptions?.format, format, logger)\n // Always explicit: the documented server default (40000) is not one of the\n // rates the endpoint accepts, so relying on it is a coin flip.\n const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE\n\n const audioConfig: BytePlusTTSAudioConfig = {\n format: audioFormat,\n sample_rate: sampleRate,\n }\n if (modelOptions?.pitch_rate !== undefined) {\n audioConfig.pitch_rate = modelOptions.pitch_rate\n }\n if (modelOptions?.loudness_rate !== undefined) {\n audioConfig.loudness_rate = modelOptions.loudness_rate\n }\n if (modelOptions?.enable_subtitle !== undefined) {\n audioConfig.enable_subtitle = modelOptions.enable_subtitle\n }\n\n // An explicit `speech_rate` always wins over the derived one — it is the\n // native unit and the only way to reach the extremes precisely.\n const speechRate =\n modelOptions?.speech_rate ??\n (speed !== undefined ? toSpeechRate(speed, logger) : undefined)\n if (speechRate !== undefined) {\n audioConfig.speech_rate = speechRate\n }\n\n const body: BytePlusTTSCreateRequest = {\n model,\n [TTS_TEXT_FIELD]: text,\n // The voice belongs inside `references`, not at the top level — a\n // top-level `speaker` is silently ignored by the server.\n references: modelOptions?.references ?? [\n {\n speaker: modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,\n },\n ],\n audio_config: audioConfig,\n }\n if (modelOptions?.watermark !== undefined) {\n // The endpoint wants an object. `true` means the audible marker, which is\n // what a caller passing a boolean is asking for.\n body.watermark =\n typeof modelOptions.watermark === 'boolean'\n ? { aigc_watermark: modelOptions.watermark }\n : modelOptions.watermark\n }\n\n return { body, audioFormat, sampleRate }\n}\n\n/**\n * Convert the cross-provider `TTSOptions.speed` multiplier into Seed Speech's\n * `speech_rate` percentage.\n *\n * ```\n * speech_rate = clamp(round((speed - 1) * 100), -50, 100)\n * ```\n *\n * so `0.5 → -50`, `1.0 → 0`, `1.5 → 50`, `2.0 → 100`.\n *\n * The range `-50`..`100` and both multiplier anchors (`-50` = 0.5×,\n * `100` = 2×) are documented, and hold on both the `/tts/create` and\n * `/tts/unidirectional` endpoints.\n *\n * `TTSOptions.speed` spans a wider 0.25×–4× than the endpoint supports, so\n * anything outside 0.5×–2× clamps (and warns) rather than erroring.\n */\nexport function toSpeechRate(speed: number, logger?: InternalLogger): number {\n const rate = Math.round((speed - 1) * 100)\n const clamped = Math.min(100, Math.max(-50, rate))\n if (clamped !== rate) {\n logger?.warn(\n `Speed ${speed}× is outside the range BytePlus Seed Speech documents (0.5×–2×) — clamping speech_rate from ${rate} to ${clamped}.`,\n { provider: 'byteplus', requestedSpeed: speed, speechRate: clamped },\n )\n }\n return clamped\n}\n\n/**\n * Map the cross-provider `TTSOptions.format` onto a Seed Speech output\n * format. An explicit `modelOptions.format` always wins.\n *\n * Seed Speech produces `wav`, `mp3`, `pcm` and `ogg_opus`. The generic\n * `opus` maps onto `ogg_opus`; `aac` and `flac` have no equivalent and fall\n * back to `mp3` (matching how the other non-OpenAI TTS adapters in this repo\n * handle unsupported codecs). The fallback is logged so it isn't silent.\n */\nfunction pickAudioFormat(\n override: BytePlusTTSAudioFormat | undefined,\n format: TTSOptions['format'] | undefined,\n logger: InternalLogger,\n): BytePlusTTSAudioFormat {\n if (override) return override\n if (!format) return 'mp3'\n switch (format) {\n case 'mp3':\n case 'wav':\n case 'pcm':\n return format\n case 'opus':\n return 'ogg_opus'\n case 'aac':\n case 'flac':\n logger.warn(\n `BytePlus Seed Speech does not support ${format} output — falling back to mp3. Set modelOptions.format to choose between wav, mp3, pcm and ogg_opus.`,\n { provider: 'byteplus', requestedFormat: format },\n )\n return 'mp3'\n }\n}\n\n/**\n * Build a per-request id for the `X-Api-Request-Id` header. Falls back to the\n * package's own id generator on runtimes without `crypto.randomUUID`.\n */\nfunction newRequestId(): string {\n return globalThis.crypto?.randomUUID?.() ?? generateId('byteplus-tts')\n}\n\n/**\n * MIME type for a Seed Speech output format.\n *\n * `pcm` is raw little-endian 16-bit samples, so its media type has to carry\n * the sample rate (RFC 3551/3555). The adapter always resolves a rate, so one\n * is always available here.\n */\nexport function getContentType(\n format: BytePlusTTSAudioFormat,\n sampleRate?: number,\n): string {\n switch (format) {\n case 'mp3':\n return 'audio/mpeg'\n case 'wav':\n return 'audio/wav'\n case 'ogg_opus':\n return 'audio/ogg;codecs=opus'\n case 'pcm':\n return `audio/L16;rate=${sampleRate ?? 24000}`\n }\n}\n\n/**\n * Coerce a `duration` / `original_duration` field to a usable number of\n * seconds.\n *\n * Both are documented as float **seconds**, so this only parses the string\n * form and drops values that can't be a length (zero, negative, non-numeric).\n * Note that `duration` may legitimately exceed\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} — a clip synthesised at\n * `speech_rate: -50` runs at half speed, so up to 120 s of billed audio can\n * be delivered as up to 240 s of playback. Do not \"correct\" a large value\n * here; the cap applies to `original_duration`.\n *\n * The subtitle timings are the mixed-unit exception: those are milliseconds\n * and are passed through untouched on {@link BytePlusTTSResult.subtitle}.\n */\nexport function toDurationSeconds(\n raw: number | string | undefined,\n): number | undefined {\n const value = typeof raw === 'string' ? Number(raw) : raw\n if (value === undefined || !Number.isFinite(value) || value <= 0) {\n return undefined\n }\n return value\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter with an explicit API key.\n *\n * The key is the **Seed Speech** key, not the Ark key used by the chat, image\n * and video adapters.\n *\n * @example\n * ```ts\n * const adapter = createBytePlusSpeech('seed-audio-1.0', process.env.BYTEPLUS_VOICE_API_KEY!)\n * ```\n */\nexport function createBytePlusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n apiKey: string,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return new BytePlusTTSAdapter(model, { ...config, apiKey })\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter, reading the API key from\n * `BYTEPLUS_VOICE_API_KEY`.\n *\n * @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.\n */\nexport function byteplusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config)\n}\n"],"mappings":";;;;;;AA2BA,IAAM,kBAAkB;;;;;;;;;;;;AAaxB,IAAM,iBAAiB;;;;;;AAOvB,IAAM,sBAAsB;;;;;;;;;;AAW5B,SAAS,WAAW,MAA4C;CAC9D,IAAI,SAAS,KAAA,GAAW,OAAO;CAC/B,OAAO,OAAO,IAAI,MAAM;AAC1B;;;;;AAMA,IAAa,+BAA+B;;;;;;;;;AAU5C,IAAa,kCAAkC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiC/C,IAAa,qBAAb,cAEU,eAAmD;CAC3D,OAAgB;CAEhB;CACA;CACA;CACA;CAEA,YAAY,OAAe,QAA6B;EACtD,MAAM,OAAO,MAAM;EACnB,MAAM,WAAW,0BAA0B,MAAM;EACjD,KAAK,SAAS,SAAS;EACvB,KAAK,UAAU,SAAS,WAAA;EACxB,KAAK,iBAAiB,SAAS,kBAAkB,CAAC;EAClD,KAAK,YAAY,SAAS,SAAS,WAAW,MAAM,KAAK,UAAU;CACrE;CAEA,MAAM,eACJ,SAC4B;EAC5B,MAAM,EAAE,QAAQ,OAAO,MAAM,OAAO,QAAQ,OAAO,iBAAiB;EAEpE,OAAO,QAAQ,mDAAmD,SAAS;GACzE,UAAU;GACV;EACF,CAAC;EAED,MAAM,EAAE,MAAM,aAAa,eAAe,oBAAoB;GAC5D;GACA;GACA;GACA;GACA;GACA;GACA;EACF,CAAC;EAED,IAAI;GACF,MAAM,WAAW,MAAM,KAAK,UAC1B,GAAG,KAAK,UAAU,mBAClB;IACE,QAAQ;IACR,SAAS,qBAAqB,KAAK,QAAQ;KACzC,GAAG,KAAK;KAIR,oBAAoB,aAAa;IACnC,CAAC;IACD,MAAM,KAAK,UAAU,IAAI;GAC3B,CACF;GAEA,MAAM,UAAU,MAAM,aAAa,QAAQ;GAE3C,IAAI,CAAC,SAAS,IACZ,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAGrE,MAAM,OAAO;GAYb,IAAI,CAAC,WAAW,KAAK,IAAI,GACvB,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAOrE,IAAI,OAAO,KAAK,UAAU,YAAY,KAAK,MAAM,WAAW,GAC1D,MAAM,IAAI,MACR,wFAC0B,MAAM,GAClC;GAGF,MAAM,WAAW,kBAAkB,KAAK,QAAQ;GAChD,MAAM,mBAAmB,kBAAkB,KAAK,iBAAiB;GAEjE,OAAO;IACL,IAAI,WAAW,KAAK,IAAI;IACxB;IACA,OAAO,KAAK;IACZ,QAAQ;IACR,aAAa,eAAe,aAAa,UAAU;IACnD,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;IACzC,GAAI,qBAAqB,KAAA,KAAa,EAAE,iBAAiB;IACzD,GAAI,KAAK,aAAa,KAAA,KAAa,EAAE,UAAU,KAAK,SAAS;IAC7D,GAAI,KAAK,QAAQ,KAAA,KAAa,EAAE,KAAK,KAAK,IAAI;GAChD;EACF,SAAS,OAAO;GACd,OAAO,OAAO,iCAAiC;IAC7C,OAAO,kBAAkB,OAAO,gCAAgC;IAChE,QAAQ;GACV,CAAC;GACD,MAAM;EACR;CACF;AACF;;;;;;;;;AAUA,SAAgB,oBAAoB,SAYlC;CACA,MAAM,EAAE,OAAO,MAAM,OAAO,QAAQ,OAAO,cAAc,WAAW;CAEpE,MAAM,cAAc,gBAAgB,cAAc,QAAQ,QAAQ,MAAM;CAGxE,MAAM,aAAa,cAAc,eAAe;CAEhD,MAAM,cAAsC;EAC1C,QAAQ;EACR,aAAa;CACf;CACA,IAAI,cAAc,eAAe,KAAA,GAC/B,YAAY,aAAa,aAAa;CAExC,IAAI,cAAc,kBAAkB,KAAA,GAClC,YAAY,gBAAgB,aAAa;CAE3C,IAAI,cAAc,oBAAoB,KAAA,GACpC,YAAY,kBAAkB,aAAa;CAK7C,MAAM,aACJ,cAAc,gBACb,UAAU,KAAA,IAAY,aAAa,OAAO,MAAM,IAAI,KAAA;CACvD,IAAI,eAAe,KAAA,GACjB,YAAY,cAAc;CAG5B,MAAM,OAAiC;EACrC;GACC,iBAAiB;EAGlB,YAAY,cAAc,cAAc,CACtC,EACE,SAAS,cAAc,WAAW,SAAA,iCACpC,CACF;EACA,cAAc;CAChB;CACA,IAAI,cAAc,cAAc,KAAA,GAG9B,KAAK,YACH,OAAO,aAAa,cAAc,YAC9B,EAAE,gBAAgB,aAAa,UAAU,IACzC,aAAa;CAGrB,OAAO;EAAE;EAAM;EAAa;CAAW;AACzC;;;;;;;;;;;;;;;;;;AAmBA,SAAgB,aAAa,OAAe,QAAiC;CAC3E,MAAM,OAAO,KAAK,OAAO,QAAQ,KAAK,GAAG;CACzC,MAAM,UAAU,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC;CACjD,IAAI,YAAY,MACd,QAAQ,KACN,SAAS,MAAM,8FAA8F,KAAK,MAAM,QAAQ,IAChI;EAAE,UAAU;EAAY,gBAAgB;EAAO,YAAY;CAAQ,CACrE;CAEF,OAAO;AACT;;;;;;;;;;AAWA,SAAS,gBACP,UACA,QACA,QACwB;CACxB,IAAI,UAAU,OAAO;CACrB,IAAI,CAAC,QAAQ,OAAO;CACpB,QAAQ,QAAR;EACE,KAAK;EACL,KAAK;EACL,KAAK,OACH,OAAO;EACT,KAAK,QACH,OAAO;EACT,KAAK;EACL,KAAK;GACH,OAAO,KACL,yCAAyC,OAAO,uGAChD;IAAE,UAAU;IAAY,iBAAiB;GAAO,CAClD;GACA,OAAO;CACX;AACF;;;;;AAMA,SAAS,eAAuB;CAC9B,OAAO,WAAW,QAAQ,aAAa,KAAK,WAAW,cAAc;AACvE;;;;;;;;AASA,SAAgB,eACd,QACA,YACQ;CACR,QAAQ,QAAR;EACE,KAAK,OACH,OAAO;EACT,KAAK,OACH,OAAO;EACT,KAAK,YACH,OAAO;EACT,KAAK,OACH,OAAO,kBAAkB,cAAc;CAC3C;AACF;;;;;;;;;;;;;;;;AAiBA,SAAgB,kBACd,KACoB;CACpB,MAAM,QAAQ,OAAO,QAAQ,WAAW,OAAO,GAAG,IAAI;CACtD,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,GAC7D;CAEF,OAAO;AACT;;;;;;;;;;;;AAaA,SAAgB,qBAGd,OACA,QACA,QAC4B;CAC5B,OAAO,IAAI,mBAAmB,OAAO;EAAE,GAAG;EAAQ;CAAO,CAAC;AAC5D;;;;;;;AAQA,SAAgB,eAGd,OACA,QAC4B;CAC5B,OAAO,qBAAqB,OAAO,8BAA8B,GAAG,MAAM;AAC5E"}
|
|
1
|
+
{"version":3,"file":"tts.js","names":[],"sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport { generateId } from '@tanstack/ai-utils'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport {\n BYTEPLUS_VOICE_BASE_URL,\n bytePlusVoiceError,\n bytePlusVoiceHeaders,\n getBytePlusVoiceApiKeyFromEnv,\n readJsonBody,\n withBytePlusVoiceDefaults,\n} from '../utils/client'\nimport type {\n TTSAlignment,\n TTSCapabilities,\n TTSOptions,\n TTSSegment,\n TTSTurn,\n} from '@tanstack/ai'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { BytePlusVoiceConfig } from '../utils/client'\nimport type { BytePlusTTSModel } from '../model-meta'\nimport type {\n BytePlusTTSAudioConfig,\n BytePlusTTSAudioFormat,\n BytePlusTTSCreateRequest,\n BytePlusTTSCreateResponse,\n BytePlusTTSReference,\n BytePlusTTSSubtitle,\n BytePlusTTSSubtitleEntry,\n} from '../audio/wire-types'\nimport type {\n BytePlusTTSProviderOptions,\n BytePlusTTSResult,\n} from '../audio/tts-provider-options'\n\n/** Path of the synchronous Seed Speech synthesis endpoint. */\nconst TTS_CREATE_PATH = '/api/v3/tts/create'\n\n/**\n * Name of the request field carrying the text to speak.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** The endpoint\n * schema (docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01) lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`; there is no request-side `text`. The `text` spelling belongs\n * to the *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits\n * under `req_params.text`. The name stays isolated here so the two spellings\n * never get conflated.\n */\nconst TTS_TEXT_FIELD = 'text_prompt' satisfies keyof BytePlusTTSCreateRequest\n\n/**\n * Sample rate used when the caller doesn't pick one. The endpoint documents a\n * default of 40000, which is not among the rates it accepts — so the adapter\n * never relies on the server default and always sends this instead.\n */\nconst DEFAULT_SAMPLE_RATE = 24000\n\n/**\n * True when a Seed Speech envelope's `code` means success.\n *\n * Seed Speech uses `0` for success and a flat numeric code otherwise\n * (`45000010 Invalid X-Api-Key`). The success envelope has not been confirmed\n * against a live key, so `code` is accepted as a number, its string form, or\n * absent — an envelope that omits `code` entirely is treated as success, which\n * is what the HTTP status already told us.\n */\nfunction isZeroCode(code: number | string | undefined): boolean {\n if (code === undefined) return true\n return Number(code) === 0\n}\n\n/**\n * Voice used when neither `TTSOptions.voice` nor `modelOptions.speaker` is\n * set — the English female \"Stokie\" voice from the TTS 2.0 generation.\n */\nexport const BYTEPLUS_DEFAULT_TTS_SPEAKER = 'en_female_stokie_uranus_bigtts'\n\n/**\n * Hard cap on a single Seed Speech synthesis, in seconds. Longer scripts must\n * be split across calls and stitched client-side.\n *\n * The cap applies to the *pre-rate* length the service bills on — the\n * `original_duration` it returns. The delivered clip can run longer than this\n * when `speech_rate` slows it down.\n */\nexport const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120\n\n/**\n * Hard cap on the `references` array, and therefore on the number of distinct\n * voices one dialogue request can use. The markers that cite them in\n * `text_prompt` run `@Audio1`..`@Audio3`.\n */\nexport const BYTEPLUS_TTS_MAX_REFERENCES = 3\n\n/**\n * BytePlus Seed Speech text-to-speech adapter.\n *\n * Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.\n * Two things differ from the Ark-hosted adapters in this package:\n *\n * - **Separate API key.** Seed Speech authenticates with `X-Api-Key` and its\n * own key (`BYTEPLUS_VOICE_API_KEY`). An `ARK_API_KEY` sent here is\n * rejected with `45000010 Invalid X-Api-Key`.\n * - **120 s output cap.** A single call synthesises at most\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} seconds of audio; longer scripts\n * have to be split and stitched client-side. The cap is measured before\n * `speech_rate` is applied, so a slowed clip can play for longer than that\n * — `result.duration` is the delivered length and `result.originalDuration`\n * is the metered one.\n *\n * Streaming synthesis (`tts/unidirectional` and the WebSocket API) is not\n * covered by this adapter — note that endpoint spells the text field `text`,\n * while this one uses `text_prompt`.\n *\n * @example\n * ```ts\n * const adapter = byteplusSpeech('seed-audio-1.0')\n * const result = await generateSpeech({\n * adapter,\n * text: 'welcome to the guitar store',\n * voice: 'en_female_stokie_uranus_bigtts',\n * format: 'mp3',\n * })\n * ```\n */\nexport class BytePlusTTSAdapter<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {\n readonly name = 'byteplus' as const\n\n /**\n * Seed Audio 1.0 does multi-role dialogue in one pass, addressed through\n * `references` (capped at 3 entries), and returns word/sentence timings\n * when `audio_config.enable_subtitle` is set.\n */\n override readonly capabilities: TTSCapabilities = {\n maxSpeakers: BYTEPLUS_TTS_MAX_REFERENCES,\n timestamps: true,\n }\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n private readonly fetchImpl: typeof fetch\n\n constructor(model: TModel, config: BytePlusVoiceConfig) {\n super(model, config)\n const resolved = withBytePlusVoiceDefaults(config)\n this.apiKey = resolved.apiKey\n this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL\n this.defaultHeaders = resolved.defaultHeaders ?? {}\n this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)\n }\n\n async generateSpeech(\n options: TTSOptions<BytePlusTTSProviderOptions>,\n ): Promise<BytePlusTTSResult> {\n const { logger, model, text, voice, format, speed, modelOptions, turns } =\n options\n\n logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {\n provider: 'byteplus',\n model,\n })\n\n const { body, audioFormat, sampleRate } = buildTTSRequestBody({\n model,\n text,\n turns,\n voice,\n format,\n speed,\n modelOptions,\n timestamps: options.timestamps,\n logger,\n })\n\n try {\n const response = await this.fetchImpl(\n `${this.baseURL}${TTS_CREATE_PATH}`,\n {\n method: 'POST',\n headers: bytePlusVoiceHeaders(this.apiKey, {\n ...this.defaultHeaders,\n // Client-generated per-request id. BytePlus echoes it in their\n // request logs, which is what support asks for when diagnosing a\n // synthesis failure.\n 'X-Api-Request-Id': newRequestId(),\n }),\n body: JSON.stringify(body),\n },\n )\n\n const payload = await readJsonBody(response)\n\n if (!response.ok) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n const data = payload as BytePlusTTSCreateResponse\n\n // Seed Speech reports status in the body, not only in the HTTP status:\n // a 200 can carry a non-zero `code`. Check it before looking at `audio`,\n // because a failed call may still return a partial or placeholder\n // payload that would otherwise be handed back as if it were valid.\n //\n // `code` is accepted as a number *or* a string. The success envelope was\n // never confirmed against a live key (no voice key yet — see\n // `audio/wire-types.ts`), and `readStringField` already tolerates both\n // forms when rendering the error, so requiring a number here would let\n // `{\"code\": \"45000010\"}` through both this gate and the one below.\n if (!isZeroCode(data.code)) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n // Belt and braces for a 200 that reports success but carries nothing to\n // play. Say that the adapter rejected it, rather than reusing the\n // envelope-error phrasing — a bare \"failed (200)\" gives no hint that the\n // response was well-formed and simply empty.\n if (typeof data.audio !== 'string' || data.audio.length === 0) {\n throw new Error(\n `BytePlus Seed Speech text-to-speech returned a success response ` +\n `with no audio (model ${model}).`,\n )\n }\n\n const duration = toDurationSeconds(data.duration)\n const originalDuration = toDurationSeconds(data.original_duration)\n\n return {\n id: generateId(this.name),\n model,\n audio: data.audio,\n format: audioFormat,\n contentType: getContentType(audioFormat, sampleRate),\n ...(duration !== undefined && { duration }),\n ...(originalDuration !== undefined && { originalDuration }),\n ...(data.subtitle !== undefined && { subtitle: data.subtitle }),\n ...toAlignmentFields(data.subtitle),\n ...(data.url !== undefined && { url: data.url }),\n }\n } catch (error) {\n logger.errors('byteplus.generateSpeech fatal', {\n error: toRunErrorPayload(error, 'byteplus.generateSpeech failed'),\n source: 'byteplus.generateSpeech',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the JSON body for `POST /api/v3/tts/create`, resolving the voice,\n * output format and rate fields in one place.\n *\n * Returns the request `body`, the resolved `audioFormat` and the\n * `sampleRate`, which the caller reports on the result and turns into a\n * `contentType`.\n */\nexport function buildTTSRequestBody(options: {\n model: string\n text: string\n /** Dialogue turns, when the caller asked for multi-role synthesis. */\n turns?: Array<TTSTurn>\n voice: string | undefined\n format: TTSOptions['format'] | undefined\n speed: number | undefined\n modelOptions: BytePlusTTSProviderOptions | undefined\n /** Core `timestamps: true` — turns on `audio_config.enable_subtitle`. */\n timestamps?: boolean\n logger: InternalLogger\n}): {\n body: BytePlusTTSCreateRequest\n audioFormat: BytePlusTTSAudioFormat\n sampleRate: number\n} {\n const { model, text, turns, voice, format, speed, modelOptions, logger } =\n options\n\n const audioFormat = pickAudioFormat(modelOptions?.format, format, logger)\n // Always explicit: the documented server default (40000) is not one of the\n // rates the endpoint accepts, so relying on it is a coin flip.\n const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE\n\n const audioConfig: BytePlusTTSAudioConfig = {\n format: audioFormat,\n sample_rate: sampleRate,\n }\n if (modelOptions?.pitch_rate !== undefined) {\n audioConfig.pitch_rate = modelOptions.pitch_rate\n }\n if (modelOptions?.loudness_rate !== undefined) {\n audioConfig.loudness_rate = modelOptions.loudness_rate\n }\n // `modelOptions.enable_subtitle` still wins, so an explicit `false` can\n // opt out of the flag even when core asked for timestamps. A `timestamps`\n // the caller never set stays off the body entirely.\n const enableSubtitle =\n modelOptions?.enable_subtitle ?? (options.timestamps || undefined)\n if (enableSubtitle !== undefined) {\n audioConfig.enable_subtitle = enableSubtitle\n }\n\n // An explicit `speech_rate` always wins over the derived one — it is the\n // native unit and the only way to reach the extremes precisely.\n const speechRate =\n modelOptions?.speech_rate ??\n (speed !== undefined ? toSpeechRate(speed, logger) : undefined)\n if (speechRate !== undefined) {\n audioConfig.speech_rate = speechRate\n }\n\n const dialogue = turns ? buildDialoguePrompt(turns) : undefined\n\n const body: BytePlusTTSCreateRequest = {\n model,\n [TTS_TEXT_FIELD]: dialogue?.textPrompt ?? text,\n // The voice belongs inside `references`, not at the top level — a\n // top-level `speaker` is silently ignored by the server.\n references: modelOptions?.references ??\n dialogue?.references ?? [\n {\n speaker:\n modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,\n },\n ],\n audio_config: audioConfig,\n }\n if (modelOptions?.watermark !== undefined) {\n // The endpoint wants an object. `true` means the audible marker, which is\n // what a caller passing a boolean is asking for.\n body.watermark =\n typeof modelOptions.watermark === 'boolean'\n ? { aigc_watermark: modelOptions.watermark }\n : modelOptions.watermark\n }\n\n return { body, audioFormat, sampleRate }\n}\n\n/**\n * Convert the cross-provider `TTSOptions.speed` multiplier into Seed Speech's\n * `speech_rate` percentage.\n *\n * ```\n * speech_rate = clamp(round((speed - 1) * 100), -50, 100)\n * ```\n *\n * so `0.5 → -50`, `1.0 → 0`, `1.5 → 50`, `2.0 → 100`.\n *\n * The range `-50`..`100` and both multiplier anchors (`-50` = 0.5×,\n * `100` = 2×) are documented, and hold on both the `/tts/create` and\n * `/tts/unidirectional` endpoints.\n *\n * `TTSOptions.speed` spans a wider 0.25×–4× than the endpoint supports, so\n * anything outside 0.5×–2× clamps (and warns) rather than erroring.\n */\nexport function toSpeechRate(speed: number, logger?: InternalLogger): number {\n const rate = Math.round((speed - 1) * 100)\n const clamped = Math.min(100, Math.max(-50, rate))\n if (clamped !== rate) {\n logger?.warn(\n `Speed ${speed}× is outside the range BytePlus Seed Speech documents (0.5×–2×) — clamping speech_rate from ${rate} to ${clamped}.`,\n { provider: 'byteplus', requestedSpeed: speed, speechRate: clamped },\n )\n }\n return clamped\n}\n\n/**\n * Map the cross-provider `TTSOptions.format` onto a Seed Speech output\n * format. An explicit `modelOptions.format` always wins.\n *\n * Seed Speech produces `wav`, `mp3`, `pcm` and `ogg_opus`. The generic\n * `opus` maps onto `ogg_opus`; `aac` and `flac` have no equivalent and fall\n * back to `mp3` (matching how the other non-OpenAI TTS adapters in this repo\n * handle unsupported codecs). The fallback is logged so it isn't silent.\n */\nfunction pickAudioFormat(\n override: BytePlusTTSAudioFormat | undefined,\n format: TTSOptions['format'] | undefined,\n logger: InternalLogger,\n): BytePlusTTSAudioFormat {\n if (override) return override\n if (!format) return 'mp3'\n switch (format) {\n case 'mp3':\n case 'wav':\n case 'pcm':\n return format\n case 'opus':\n return 'ogg_opus'\n case 'aac':\n case 'flac':\n logger.warn(\n `BytePlus Seed Speech does not support ${format} output — falling back to mp3. Set modelOptions.format to choose between wav, mp3, pcm and ogg_opus.`,\n { provider: 'byteplus', requestedFormat: format },\n )\n return 'mp3'\n }\n}\n\n/**\n * Build a per-request id for the `X-Api-Request-Id` header. Falls back to the\n * package's own id generator on runtimes without `crypto.randomUUID`.\n */\nfunction newRequestId(): string {\n return globalThis.crypto?.randomUUID?.() ?? generateId('byteplus-tts')\n}\n\n/**\n * MIME type for a Seed Speech output format.\n *\n * `pcm` is raw little-endian 16-bit samples, so its media type has to carry\n * the sample rate (RFC 3551/3555). The adapter always resolves a rate, so one\n * is always available here.\n */\nexport function getContentType(\n format: BytePlusTTSAudioFormat,\n sampleRate?: number,\n): string {\n switch (format) {\n case 'mp3':\n return 'audio/mpeg'\n case 'wav':\n return 'audio/wav'\n case 'ogg_opus':\n return 'audio/ogg;codecs=opus'\n case 'pcm':\n return `audio/L16;rate=${sampleRate ?? 24000}`\n }\n}\n\n/**\n * Turn dialogue turns into the one-prompt form Seed Audio 1.0 takes: each\n * distinct voice becomes a `references` entry, and every line of the script\n * cites its voice by the reference's 1-based position (`@Audio1`..`@Audio3`),\n * which is the marker convention the reference array already uses.\n *\n * **The prompt form is unprobed.** The flat `references[]` member shape is\n * confirmed (see `BytePlusTTSReference`), and the model card documents\n * multi-role dialogue and positional reference markers, but no worked\n * multi-speaker example was available and no Seed Speech key existed to\n * confirm how lines should cite their voice.\n */\nfunction buildDialoguePrompt(turns: Array<TTSTurn>): {\n textPrompt: string\n references: Array<BytePlusTTSReference>\n} {\n const voices = [...new Set(turns.map((turn) => turn.voice))]\n return {\n textPrompt: turns\n .map((turn) => `@Audio${voices.indexOf(turn.voice) + 1}: ${turn.text}`)\n .join('\\n'),\n references: voices.map((speaker) => ({ speaker })),\n }\n}\n\n/**\n * Map the BytePlus `subtitle` block onto the cross-provider `alignment` /\n * `segments` fields.\n *\n * `words` is the finest granularity Seed Speech reports, so it becomes\n * `alignment`; `sentences` is utterance segmentation, so it becomes\n * `segments`. **Subtitle times are milliseconds** while the rest of the\n * response is seconds — the conversion happens here, once.\n *\n * The raw block stays on {@link BytePlusTTSResult.subtitle} for callers\n * already reading it.\n */\nfunction toAlignmentFields(subtitle: BytePlusTTSSubtitle | undefined): {\n alignment?: TTSAlignment\n segments?: Array<TTSSegment>\n} {\n if (!subtitle) return {}\n const fields: { alignment?: TTSAlignment; segments?: Array<TTSSegment> } = {}\n const words = subtitle.words?.filter(isTimedEntry)\n if (words && words.length > 0) {\n fields.alignment = {\n unit: 'word',\n texts: words.map((word) => word.text ?? ''),\n startSeconds: words.map((word) => word.start_time / 1000),\n endSeconds: words.map((word) => word.end_time / 1000),\n }\n }\n const sentences = subtitle.sentences?.filter(isTimedEntry)\n if (sentences && sentences.length > 0) {\n fields.segments = sentences.map((sentence) => ({\n startSeconds: sentence.start_time / 1000,\n endSeconds: sentence.end_time / 1000,\n ...(sentence.text !== undefined && { text: sentence.text }),\n }))\n }\n return fields\n}\n\n/** Both times present, so the entry can be placed on the timeline. */\nfunction isTimedEntry(\n entry: BytePlusTTSSubtitleEntry,\n): entry is BytePlusTTSSubtitleEntry & {\n start_time: number\n end_time: number\n} {\n return (\n typeof entry.start_time === 'number' && typeof entry.end_time === 'number'\n )\n}\n\n/**\n * Coerce a `duration` / `original_duration` field to a usable number of\n * seconds.\n *\n * Both are documented as float **seconds**, so this only parses the string\n * form and drops values that can't be a length (zero, negative, non-numeric).\n * Note that `duration` may legitimately exceed\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} — a clip synthesised at\n * `speech_rate: -50` runs at half speed, so up to 120 s of billed audio can\n * be delivered as up to 240 s of playback. Do not \"correct\" a large value\n * here; the cap applies to `original_duration`.\n *\n * The subtitle timings are the mixed-unit exception: those are milliseconds\n * and are passed through untouched on {@link BytePlusTTSResult.subtitle}.\n */\nexport function toDurationSeconds(\n raw: number | string | undefined,\n): number | undefined {\n const value = typeof raw === 'string' ? Number(raw) : raw\n if (value === undefined || !Number.isFinite(value) || value <= 0) {\n return undefined\n }\n return value\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter with an explicit API key.\n *\n * The key is the **Seed Speech** key, not the Ark key used by the chat, image\n * and video adapters.\n *\n * @example\n * ```ts\n * const adapter = createBytePlusSpeech('seed-audio-1.0', process.env.BYTEPLUS_VOICE_API_KEY!)\n * ```\n */\nexport function createBytePlusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n apiKey: string,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return new BytePlusTTSAdapter(model, { ...config, apiKey })\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter, reading the API key from\n * `BYTEPLUS_VOICE_API_KEY`.\n *\n * @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.\n */\nexport function byteplusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config)\n}\n"],"mappings":";;;;;;AAoCA,IAAM,kBAAkB;;;;;;;;;;;;AAaxB,IAAM,iBAAiB;;;;;;AAOvB,IAAM,sBAAsB;;;;;;;;;;AAW5B,SAAS,WAAW,MAA4C;CAC9D,IAAI,SAAS,KAAA,GAAW,OAAO;CAC/B,OAAO,OAAO,IAAI,MAAM;AAC1B;;;;;AAMA,IAAa,+BAA+B;;;;;;;;;AAU5C,IAAa,kCAAkC;;;;;;AAO/C,IAAa,8BAA8B;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiC3C,IAAa,qBAAb,cAEU,eAAmD;CAC3D,OAAgB;;;;;;CAOhB,eAAkD;EAChD,aAAA;EACA,YAAY;CACd;CAEA;CACA;CACA;CACA;CAEA,YAAY,OAAe,QAA6B;EACtD,MAAM,OAAO,MAAM;EACnB,MAAM,WAAW,0BAA0B,MAAM;EACjD,KAAK,SAAS,SAAS;EACvB,KAAK,UAAU,SAAS,WAAA;EACxB,KAAK,iBAAiB,SAAS,kBAAkB,CAAC;EAClD,KAAK,YAAY,SAAS,SAAS,WAAW,MAAM,KAAK,UAAU;CACrE;CAEA,MAAM,eACJ,SAC4B;EAC5B,MAAM,EAAE,QAAQ,OAAO,MAAM,OAAO,QAAQ,OAAO,cAAc,UAC/D;EAEF,OAAO,QAAQ,mDAAmD,SAAS;GACzE,UAAU;GACV;EACF,CAAC;EAED,MAAM,EAAE,MAAM,aAAa,eAAe,oBAAoB;GAC5D;GACA;GACA;GACA;GACA;GACA;GACA;GACA,YAAY,QAAQ;GACpB;EACF,CAAC;EAED,IAAI;GACF,MAAM,WAAW,MAAM,KAAK,UAC1B,GAAG,KAAK,UAAU,mBAClB;IACE,QAAQ;IACR,SAAS,qBAAqB,KAAK,QAAQ;KACzC,GAAG,KAAK;KAIR,oBAAoB,aAAa;IACnC,CAAC;IACD,MAAM,KAAK,UAAU,IAAI;GAC3B,CACF;GAEA,MAAM,UAAU,MAAM,aAAa,QAAQ;GAE3C,IAAI,CAAC,SAAS,IACZ,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAGrE,MAAM,OAAO;GAYb,IAAI,CAAC,WAAW,KAAK,IAAI,GACvB,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAOrE,IAAI,OAAO,KAAK,UAAU,YAAY,KAAK,MAAM,WAAW,GAC1D,MAAM,IAAI,MACR,wFAC0B,MAAM,GAClC;GAGF,MAAM,WAAW,kBAAkB,KAAK,QAAQ;GAChD,MAAM,mBAAmB,kBAAkB,KAAK,iBAAiB;GAEjE,OAAO;IACL,IAAI,WAAW,KAAK,IAAI;IACxB;IACA,OAAO,KAAK;IACZ,QAAQ;IACR,aAAa,eAAe,aAAa,UAAU;IACnD,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;IACzC,GAAI,qBAAqB,KAAA,KAAa,EAAE,iBAAiB;IACzD,GAAI,KAAK,aAAa,KAAA,KAAa,EAAE,UAAU,KAAK,SAAS;IAC7D,GAAG,kBAAkB,KAAK,QAAQ;IAClC,GAAI,KAAK,QAAQ,KAAA,KAAa,EAAE,KAAK,KAAK,IAAI;GAChD;EACF,SAAS,OAAO;GACd,OAAO,OAAO,iCAAiC;IAC7C,OAAO,kBAAkB,OAAO,gCAAgC;IAChE,QAAQ;GACV,CAAC;GACD,MAAM;EACR;CACF;AACF;;;;;;;;;AAUA,SAAgB,oBAAoB,SAgBlC;CACA,MAAM,EAAE,OAAO,MAAM,OAAO,OAAO,QAAQ,OAAO,cAAc,WAC9D;CAEF,MAAM,cAAc,gBAAgB,cAAc,QAAQ,QAAQ,MAAM;CAGxE,MAAM,aAAa,cAAc,eAAe;CAEhD,MAAM,cAAsC;EAC1C,QAAQ;EACR,aAAa;CACf;CACA,IAAI,cAAc,eAAe,KAAA,GAC/B,YAAY,aAAa,aAAa;CAExC,IAAI,cAAc,kBAAkB,KAAA,GAClC,YAAY,gBAAgB,aAAa;CAK3C,MAAM,iBACJ,cAAc,oBAAoB,QAAQ,cAAc,KAAA;CAC1D,IAAI,mBAAmB,KAAA,GACrB,YAAY,kBAAkB;CAKhC,MAAM,aACJ,cAAc,gBACb,UAAU,KAAA,IAAY,aAAa,OAAO,MAAM,IAAI,KAAA;CACvD,IAAI,eAAe,KAAA,GACjB,YAAY,cAAc;CAG5B,MAAM,WAAW,QAAQ,oBAAoB,KAAK,IAAI,KAAA;CAEtD,MAAM,OAAiC;EACrC;GACC,iBAAiB,UAAU,cAAc;EAG1C,YAAY,cAAc,cACxB,UAAU,cAAc,CACtB,EACE,SACE,cAAc,WAAW,SAAA,iCAC7B,CACF;EACF,cAAc;CAChB;CACA,IAAI,cAAc,cAAc,KAAA,GAG9B,KAAK,YACH,OAAO,aAAa,cAAc,YAC9B,EAAE,gBAAgB,aAAa,UAAU,IACzC,aAAa;CAGrB,OAAO;EAAE;EAAM;EAAa;CAAW;AACzC;;;;;;;;;;;;;;;;;;AAmBA,SAAgB,aAAa,OAAe,QAAiC;CAC3E,MAAM,OAAO,KAAK,OAAO,QAAQ,KAAK,GAAG;CACzC,MAAM,UAAU,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC;CACjD,IAAI,YAAY,MACd,QAAQ,KACN,SAAS,MAAM,8FAA8F,KAAK,MAAM,QAAQ,IAChI;EAAE,UAAU;EAAY,gBAAgB;EAAO,YAAY;CAAQ,CACrE;CAEF,OAAO;AACT;;;;;;;;;;AAWA,SAAS,gBACP,UACA,QACA,QACwB;CACxB,IAAI,UAAU,OAAO;CACrB,IAAI,CAAC,QAAQ,OAAO;CACpB,QAAQ,QAAR;EACE,KAAK;EACL,KAAK;EACL,KAAK,OACH,OAAO;EACT,KAAK,QACH,OAAO;EACT,KAAK;EACL,KAAK;GACH,OAAO,KACL,yCAAyC,OAAO,uGAChD;IAAE,UAAU;IAAY,iBAAiB;GAAO,CAClD;GACA,OAAO;CACX;AACF;;;;;AAMA,SAAS,eAAuB;CAC9B,OAAO,WAAW,QAAQ,aAAa,KAAK,WAAW,cAAc;AACvE;;;;;;;;AASA,SAAgB,eACd,QACA,YACQ;CACR,QAAQ,QAAR;EACE,KAAK,OACH,OAAO;EACT,KAAK,OACH,OAAO;EACT,KAAK,YACH,OAAO;EACT,KAAK,OACH,OAAO,kBAAkB,cAAc;CAC3C;AACF;;;;;;;;;;;;;AAcA,SAAS,oBAAoB,OAG3B;CACA,MAAM,SAAS,CAAC,GAAG,IAAI,IAAI,MAAM,KAAK,SAAS,KAAK,KAAK,CAAC,CAAC;CAC3D,OAAO;EACL,YAAY,MACT,KAAK,SAAS,SAAS,OAAO,QAAQ,KAAK,KAAK,IAAI,EAAE,IAAI,KAAK,MAAM,CAAC,CACtE,KAAK,IAAI;EACZ,YAAY,OAAO,KAAK,aAAa,EAAE,QAAQ,EAAE;CACnD;AACF;;;;;;;;;;;;;AAcA,SAAS,kBAAkB,UAGzB;CACA,IAAI,CAAC,UAAU,OAAO,CAAC;CACvB,MAAM,SAAqE,CAAC;CAC5E,MAAM,QAAQ,SAAS,OAAO,OAAO,YAAY;CACjD,IAAI,SAAS,MAAM,SAAS,GAC1B,OAAO,YAAY;EACjB,MAAM;EACN,OAAO,MAAM,KAAK,SAAS,KAAK,QAAQ,EAAE;EAC1C,cAAc,MAAM,KAAK,SAAS,KAAK,aAAa,GAAI;EACxD,YAAY,MAAM,KAAK,SAAS,KAAK,WAAW,GAAI;CACtD;CAEF,MAAM,YAAY,SAAS,WAAW,OAAO,YAAY;CACzD,IAAI,aAAa,UAAU,SAAS,GAClC,OAAO,WAAW,UAAU,KAAK,cAAc;EAC7C,cAAc,SAAS,aAAa;EACpC,YAAY,SAAS,WAAW;EAChC,GAAI,SAAS,SAAS,KAAA,KAAa,EAAE,MAAM,SAAS,KAAK;CAC3D,EAAE;CAEJ,OAAO;AACT;;AAGA,SAAS,aACP,OAIA;CACA,OACE,OAAO,MAAM,eAAe,YAAY,OAAO,MAAM,aAAa;AAEtE;;;;;;;;;;;;;;;;AAiBA,SAAgB,kBACd,KACoB;CACpB,MAAM,QAAQ,OAAO,QAAQ,WAAW,OAAO,GAAG,IAAI;CACtD,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,GAC7D;CAEF,OAAO;AACT;;;;;;;;;;;;AAaA,SAAgB,qBAGd,OACA,QACA,QAC4B;CAC5B,OAAO,IAAI,mBAAmB,OAAO;EAAE,GAAG;EAAQ;CAAO,CAAC;AAC5D;;;;;;;AAQA,SAAgB,eAGd,OACA,QAC4B;CAC5B,OAAO,qBAAqB,OAAO,8BAA8B,GAAG,MAAM;AAC5E"}
|
package/dist/esm/index.d.ts
CHANGED
|
@@ -3,7 +3,7 @@ export type { BytePlusVideoConfig } from './adapters/video.js';
|
|
|
3
3
|
export { parseBytePlusVideoSize, resolveBytePlusVideoResolution, resolveBytePlusVideoSize, supportsAudioOnlyReference, supportsLastFrame, supportsReferenceMedia, } from './video/video-provider-options.js';
|
|
4
4
|
export type { BytePlusVideoModelProviderOptionsByName, BytePlusVideoOutputFormat, BytePlusVideoProviderOptions, BytePlusVideoServiceTier, } from './video/video-provider-options.js';
|
|
5
5
|
export type { BytePlusVideoContentPart, BytePlusVideoContentRole, BytePlusVideoCreateRequest, BytePlusVideoCreateResponse, BytePlusVideoTask, BytePlusVideoTaskContent, BytePlusVideoTaskError, BytePlusVideoTaskListItem, BytePlusVideoTaskListResponse, BytePlusVideoTaskStatus, BytePlusVideoTaskUsage, } from './video/wire-types.js';
|
|
6
|
-
export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BytePlusTTSAdapter, byteplusSpeech, createBytePlusSpeech, toSpeechRate, } from './adapters/tts.js';
|
|
6
|
+
export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MAX_REFERENCES, BytePlusTTSAdapter, byteplusSpeech, createBytePlusSpeech, toSpeechRate, } from './adapters/tts.js';
|
|
7
7
|
export type { BytePlusTTSProviderOptions, BytePlusTTSResult, BytePlusTTSVoice, } from './audio/tts-provider-options.js';
|
|
8
8
|
export { BytePlusTranscriptionAdapter, byteplusTranscription, createBytePlusTranscription, } from './adapters/transcription.js';
|
|
9
9
|
export type { BytePlusTranscriptionWord } from './adapters/transcription.js';
|
package/dist/esm/index.js
CHANGED
|
@@ -2,10 +2,10 @@ import { BYTEPLUS_ARK_BASE_URL, BYTEPLUS_VOICE_BASE_URL, bytePlusArkError, byteP
|
|
|
2
2
|
import { BYTEPLUS_CHAT_MODELS, BYTEPLUS_IMAGE_MAX_REFERENCE_IMAGES, BYTEPLUS_IMAGE_MODELS, BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS, BYTEPLUS_THINKING_SUMMARY_MODELS, BYTEPLUS_TRANSCRIPTION_MODELS, BYTEPLUS_TTS_MODELS, BYTEPLUS_VIDEO_DURATIONS, BYTEPLUS_VIDEO_FALLBACK_DURATIONS, BYTEPLUS_VIDEO_MODELS, emitsEncryptedContent, getBytePlusVideoDurationOptions, isKnownBytePlusVideoModel, supportsStructuredOutput } from "./model-meta.js";
|
|
3
3
|
import { parseBytePlusVideoSize, resolveBytePlusVideoResolution, resolveBytePlusVideoSize, supportsAudioOnlyReference, supportsLastFrame, supportsReferenceMedia } from "./video/video-provider-options.js";
|
|
4
4
|
import { BytePlusVideoAdapter, byteplusVideo, createBytePlusVideo } from "./adapters/video.js";
|
|
5
|
-
import { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BytePlusTTSAdapter, byteplusSpeech, createBytePlusSpeech, toSpeechRate } from "./adapters/tts.js";
|
|
5
|
+
import { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MAX_REFERENCES, BytePlusTTSAdapter, byteplusSpeech, createBytePlusSpeech, toSpeechRate } from "./adapters/tts.js";
|
|
6
6
|
import { BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_TTS_SAMPLE_RATES } from "./audio/wire-types.js";
|
|
7
7
|
import { BytePlusTranscriptionAdapter, byteplusTranscription, createBytePlusTranscription } from "./adapters/transcription.js";
|
|
8
8
|
import { BYTEPLUS_IMAGE_MAX_PROMPT_WORDS, BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES, BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS, parseBytePlusImageSize } from "./image/image-provider-options.js";
|
|
9
9
|
import { BytePlusImageAdapter, byteplusImage, createBytePlusImage } from "./adapters/image.js";
|
|
10
10
|
import { BytePlusTextAdapter, byteplusText, createBytePlusText } from "./adapters/text.js";
|
|
11
|
-
export { BYTEPLUS_ARK_BASE_URL, BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_CHAT_MODELS, BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_IMAGE_MAX_PROMPT_WORDS, BYTEPLUS_IMAGE_MAX_REFERENCE_IMAGES, BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES, BYTEPLUS_IMAGE_MODELS, BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS, BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS, BYTEPLUS_THINKING_SUMMARY_MODELS, BYTEPLUS_TRANSCRIPTION_MODELS, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MODELS, BYTEPLUS_TTS_SAMPLE_RATES, BYTEPLUS_VIDEO_DURATIONS, BYTEPLUS_VIDEO_FALLBACK_DURATIONS, BYTEPLUS_VIDEO_MODELS, BYTEPLUS_VOICE_BASE_URL, BytePlusImageAdapter, BytePlusTTSAdapter, BytePlusTextAdapter, BytePlusTranscriptionAdapter, BytePlusVideoAdapter, bytePlusArkError, bytePlusArkHeaders, bytePlusVoiceError, bytePlusVoiceHeaders, byteplusImage, byteplusSpeech, byteplusText, byteplusTranscription, byteplusVideo, createBytePlusImage, createBytePlusSpeech, createBytePlusText, createBytePlusTranscription, createBytePlusVideo, emitsEncryptedContent, getBytePlusArkApiKeyFromEnv, getBytePlusVideoDurationOptions, getBytePlusVoiceApiKeyFromEnv, isKnownBytePlusVideoModel, parseBytePlusImageSize, parseBytePlusVideoSize, resolveBytePlusVideoResolution, resolveBytePlusVideoSize, supportsAudioOnlyReference, supportsLastFrame, supportsReferenceMedia, supportsStructuredOutput, toSpeechRate, withBytePlusArkDefaults, withBytePlusVoiceDefaults };
|
|
11
|
+
export { BYTEPLUS_ARK_BASE_URL, BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_CHAT_MODELS, BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_IMAGE_MAX_PROMPT_WORDS, BYTEPLUS_IMAGE_MAX_REFERENCE_IMAGES, BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES, BYTEPLUS_IMAGE_MODELS, BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS, BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS, BYTEPLUS_THINKING_SUMMARY_MODELS, BYTEPLUS_TRANSCRIPTION_MODELS, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MAX_REFERENCES, BYTEPLUS_TTS_MODELS, BYTEPLUS_TTS_SAMPLE_RATES, BYTEPLUS_VIDEO_DURATIONS, BYTEPLUS_VIDEO_FALLBACK_DURATIONS, BYTEPLUS_VIDEO_MODELS, BYTEPLUS_VOICE_BASE_URL, BytePlusImageAdapter, BytePlusTTSAdapter, BytePlusTextAdapter, BytePlusTranscriptionAdapter, BytePlusVideoAdapter, bytePlusArkError, bytePlusArkHeaders, bytePlusVoiceError, bytePlusVoiceHeaders, byteplusImage, byteplusSpeech, byteplusText, byteplusTranscription, byteplusVideo, createBytePlusImage, createBytePlusSpeech, createBytePlusText, createBytePlusTranscription, createBytePlusVideo, emitsEncryptedContent, getBytePlusArkApiKeyFromEnv, getBytePlusVideoDurationOptions, getBytePlusVoiceApiKeyFromEnv, isKnownBytePlusVideoModel, parseBytePlusImageSize, parseBytePlusVideoSize, resolveBytePlusVideoResolution, resolveBytePlusVideoSize, supportsAudioOnlyReference, supportsLastFrame, supportsReferenceMedia, supportsStructuredOutput, toSpeechRate, withBytePlusArkDefaults, withBytePlusVoiceDefaults };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai-byteplus",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.4.2",
|
|
4
4
|
"description": "BytePlus ModelArk adapter for TanStack AI: Seed LLM chat, Seedance video, Seedream image, and Seed Speech TTS/ASR.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -54,15 +54,15 @@
|
|
|
54
54
|
"devDependencies": {
|
|
55
55
|
"@vitest/coverage-v8": "4.1.10",
|
|
56
56
|
"vite": "^8.2.1",
|
|
57
|
-
"@tanstack/ai": "0.
|
|
57
|
+
"@tanstack/ai": "0.58.0"
|
|
58
58
|
},
|
|
59
59
|
"peerDependencies": {
|
|
60
|
-
"@tanstack/ai": "^0.
|
|
60
|
+
"@tanstack/ai": "^0.58.0"
|
|
61
61
|
},
|
|
62
62
|
"dependencies": {
|
|
63
63
|
"openai": "^6.41.0",
|
|
64
|
-
"@tanstack/ai-utils": "^0.4.
|
|
65
|
-
"@tanstack/openai-base": "^0.10.
|
|
64
|
+
"@tanstack/ai-utils": "^0.4.1",
|
|
65
|
+
"@tanstack/openai-base": "^0.10.15"
|
|
66
66
|
},
|
|
67
67
|
"scripts": {
|
|
68
68
|
"build": "vite build",
|
package/src/adapters/tts.ts
CHANGED
|
@@ -9,7 +9,13 @@ import {
|
|
|
9
9
|
readJsonBody,
|
|
10
10
|
withBytePlusVoiceDefaults,
|
|
11
11
|
} from '../utils/client'
|
|
12
|
-
import type {
|
|
12
|
+
import type {
|
|
13
|
+
TTSAlignment,
|
|
14
|
+
TTSCapabilities,
|
|
15
|
+
TTSOptions,
|
|
16
|
+
TTSSegment,
|
|
17
|
+
TTSTurn,
|
|
18
|
+
} from '@tanstack/ai'
|
|
13
19
|
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
14
20
|
import type { BytePlusVoiceConfig } from '../utils/client'
|
|
15
21
|
import type { BytePlusTTSModel } from '../model-meta'
|
|
@@ -18,6 +24,9 @@ import type {
|
|
|
18
24
|
BytePlusTTSAudioFormat,
|
|
19
25
|
BytePlusTTSCreateRequest,
|
|
20
26
|
BytePlusTTSCreateResponse,
|
|
27
|
+
BytePlusTTSReference,
|
|
28
|
+
BytePlusTTSSubtitle,
|
|
29
|
+
BytePlusTTSSubtitleEntry,
|
|
21
30
|
} from '../audio/wire-types'
|
|
22
31
|
import type {
|
|
23
32
|
BytePlusTTSProviderOptions,
|
|
@@ -77,6 +86,13 @@ export const BYTEPLUS_DEFAULT_TTS_SPEAKER = 'en_female_stokie_uranus_bigtts'
|
|
|
77
86
|
*/
|
|
78
87
|
export const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120
|
|
79
88
|
|
|
89
|
+
/**
|
|
90
|
+
* Hard cap on the `references` array, and therefore on the number of distinct
|
|
91
|
+
* voices one dialogue request can use. The markers that cite them in
|
|
92
|
+
* `text_prompt` run `@Audio1`..`@Audio3`.
|
|
93
|
+
*/
|
|
94
|
+
export const BYTEPLUS_TTS_MAX_REFERENCES = 3
|
|
95
|
+
|
|
80
96
|
/**
|
|
81
97
|
* BytePlus Seed Speech text-to-speech adapter.
|
|
82
98
|
*
|
|
@@ -113,6 +129,16 @@ export class BytePlusTTSAdapter<
|
|
|
113
129
|
> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {
|
|
114
130
|
readonly name = 'byteplus' as const
|
|
115
131
|
|
|
132
|
+
/**
|
|
133
|
+
* Seed Audio 1.0 does multi-role dialogue in one pass, addressed through
|
|
134
|
+
* `references` (capped at 3 entries), and returns word/sentence timings
|
|
135
|
+
* when `audio_config.enable_subtitle` is set.
|
|
136
|
+
*/
|
|
137
|
+
override readonly capabilities: TTSCapabilities = {
|
|
138
|
+
maxSpeakers: BYTEPLUS_TTS_MAX_REFERENCES,
|
|
139
|
+
timestamps: true,
|
|
140
|
+
}
|
|
141
|
+
|
|
116
142
|
private readonly apiKey: string
|
|
117
143
|
private readonly baseURL: string
|
|
118
144
|
private readonly defaultHeaders: Record<string, string>
|
|
@@ -130,7 +156,8 @@ export class BytePlusTTSAdapter<
|
|
|
130
156
|
async generateSpeech(
|
|
131
157
|
options: TTSOptions<BytePlusTTSProviderOptions>,
|
|
132
158
|
): Promise<BytePlusTTSResult> {
|
|
133
|
-
const { logger, model, text, voice, format, speed, modelOptions } =
|
|
159
|
+
const { logger, model, text, voice, format, speed, modelOptions, turns } =
|
|
160
|
+
options
|
|
134
161
|
|
|
135
162
|
logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {
|
|
136
163
|
provider: 'byteplus',
|
|
@@ -140,10 +167,12 @@ export class BytePlusTTSAdapter<
|
|
|
140
167
|
const { body, audioFormat, sampleRate } = buildTTSRequestBody({
|
|
141
168
|
model,
|
|
142
169
|
text,
|
|
170
|
+
turns,
|
|
143
171
|
voice,
|
|
144
172
|
format,
|
|
145
173
|
speed,
|
|
146
174
|
modelOptions,
|
|
175
|
+
timestamps: options.timestamps,
|
|
147
176
|
logger,
|
|
148
177
|
})
|
|
149
178
|
|
|
@@ -208,6 +237,7 @@ export class BytePlusTTSAdapter<
|
|
|
208
237
|
...(duration !== undefined && { duration }),
|
|
209
238
|
...(originalDuration !== undefined && { originalDuration }),
|
|
210
239
|
...(data.subtitle !== undefined && { subtitle: data.subtitle }),
|
|
240
|
+
...toAlignmentFields(data.subtitle),
|
|
211
241
|
...(data.url !== undefined && { url: data.url }),
|
|
212
242
|
}
|
|
213
243
|
} catch (error) {
|
|
@@ -231,17 +261,22 @@ export class BytePlusTTSAdapter<
|
|
|
231
261
|
export function buildTTSRequestBody(options: {
|
|
232
262
|
model: string
|
|
233
263
|
text: string
|
|
264
|
+
/** Dialogue turns, when the caller asked for multi-role synthesis. */
|
|
265
|
+
turns?: Array<TTSTurn>
|
|
234
266
|
voice: string | undefined
|
|
235
267
|
format: TTSOptions['format'] | undefined
|
|
236
268
|
speed: number | undefined
|
|
237
269
|
modelOptions: BytePlusTTSProviderOptions | undefined
|
|
270
|
+
/** Core `timestamps: true` — turns on `audio_config.enable_subtitle`. */
|
|
271
|
+
timestamps?: boolean
|
|
238
272
|
logger: InternalLogger
|
|
239
273
|
}): {
|
|
240
274
|
body: BytePlusTTSCreateRequest
|
|
241
275
|
audioFormat: BytePlusTTSAudioFormat
|
|
242
276
|
sampleRate: number
|
|
243
277
|
} {
|
|
244
|
-
const { model, text, voice, format, speed, modelOptions, logger } =
|
|
278
|
+
const { model, text, turns, voice, format, speed, modelOptions, logger } =
|
|
279
|
+
options
|
|
245
280
|
|
|
246
281
|
const audioFormat = pickAudioFormat(modelOptions?.format, format, logger)
|
|
247
282
|
// Always explicit: the documented server default (40000) is not one of the
|
|
@@ -258,8 +293,13 @@ export function buildTTSRequestBody(options: {
|
|
|
258
293
|
if (modelOptions?.loudness_rate !== undefined) {
|
|
259
294
|
audioConfig.loudness_rate = modelOptions.loudness_rate
|
|
260
295
|
}
|
|
261
|
-
|
|
262
|
-
|
|
296
|
+
// `modelOptions.enable_subtitle` still wins, so an explicit `false` can
|
|
297
|
+
// opt out of the flag even when core asked for timestamps. A `timestamps`
|
|
298
|
+
// the caller never set stays off the body entirely.
|
|
299
|
+
const enableSubtitle =
|
|
300
|
+
modelOptions?.enable_subtitle ?? (options.timestamps || undefined)
|
|
301
|
+
if (enableSubtitle !== undefined) {
|
|
302
|
+
audioConfig.enable_subtitle = enableSubtitle
|
|
263
303
|
}
|
|
264
304
|
|
|
265
305
|
// An explicit `speech_rate` always wins over the derived one — it is the
|
|
@@ -271,16 +311,20 @@ export function buildTTSRequestBody(options: {
|
|
|
271
311
|
audioConfig.speech_rate = speechRate
|
|
272
312
|
}
|
|
273
313
|
|
|
314
|
+
const dialogue = turns ? buildDialoguePrompt(turns) : undefined
|
|
315
|
+
|
|
274
316
|
const body: BytePlusTTSCreateRequest = {
|
|
275
317
|
model,
|
|
276
|
-
[TTS_TEXT_FIELD]: text,
|
|
318
|
+
[TTS_TEXT_FIELD]: dialogue?.textPrompt ?? text,
|
|
277
319
|
// The voice belongs inside `references`, not at the top level — a
|
|
278
320
|
// top-level `speaker` is silently ignored by the server.
|
|
279
|
-
references: modelOptions?.references ??
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
321
|
+
references: modelOptions?.references ??
|
|
322
|
+
dialogue?.references ?? [
|
|
323
|
+
{
|
|
324
|
+
speaker:
|
|
325
|
+
modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,
|
|
326
|
+
},
|
|
327
|
+
],
|
|
284
328
|
audio_config: audioConfig,
|
|
285
329
|
}
|
|
286
330
|
if (modelOptions?.watermark !== undefined) {
|
|
@@ -388,6 +432,81 @@ export function getContentType(
|
|
|
388
432
|
}
|
|
389
433
|
}
|
|
390
434
|
|
|
435
|
+
/**
|
|
436
|
+
* Turn dialogue turns into the one-prompt form Seed Audio 1.0 takes: each
|
|
437
|
+
* distinct voice becomes a `references` entry, and every line of the script
|
|
438
|
+
* cites its voice by the reference's 1-based position (`@Audio1`..`@Audio3`),
|
|
439
|
+
* which is the marker convention the reference array already uses.
|
|
440
|
+
*
|
|
441
|
+
* **The prompt form is unprobed.** The flat `references[]` member shape is
|
|
442
|
+
* confirmed (see `BytePlusTTSReference`), and the model card documents
|
|
443
|
+
* multi-role dialogue and positional reference markers, but no worked
|
|
444
|
+
* multi-speaker example was available and no Seed Speech key existed to
|
|
445
|
+
* confirm how lines should cite their voice.
|
|
446
|
+
*/
|
|
447
|
+
function buildDialoguePrompt(turns: Array<TTSTurn>): {
|
|
448
|
+
textPrompt: string
|
|
449
|
+
references: Array<BytePlusTTSReference>
|
|
450
|
+
} {
|
|
451
|
+
const voices = [...new Set(turns.map((turn) => turn.voice))]
|
|
452
|
+
return {
|
|
453
|
+
textPrompt: turns
|
|
454
|
+
.map((turn) => `@Audio${voices.indexOf(turn.voice) + 1}: ${turn.text}`)
|
|
455
|
+
.join('\n'),
|
|
456
|
+
references: voices.map((speaker) => ({ speaker })),
|
|
457
|
+
}
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
/**
|
|
461
|
+
* Map the BytePlus `subtitle` block onto the cross-provider `alignment` /
|
|
462
|
+
* `segments` fields.
|
|
463
|
+
*
|
|
464
|
+
* `words` is the finest granularity Seed Speech reports, so it becomes
|
|
465
|
+
* `alignment`; `sentences` is utterance segmentation, so it becomes
|
|
466
|
+
* `segments`. **Subtitle times are milliseconds** while the rest of the
|
|
467
|
+
* response is seconds — the conversion happens here, once.
|
|
468
|
+
*
|
|
469
|
+
* The raw block stays on {@link BytePlusTTSResult.subtitle} for callers
|
|
470
|
+
* already reading it.
|
|
471
|
+
*/
|
|
472
|
+
function toAlignmentFields(subtitle: BytePlusTTSSubtitle | undefined): {
|
|
473
|
+
alignment?: TTSAlignment
|
|
474
|
+
segments?: Array<TTSSegment>
|
|
475
|
+
} {
|
|
476
|
+
if (!subtitle) return {}
|
|
477
|
+
const fields: { alignment?: TTSAlignment; segments?: Array<TTSSegment> } = {}
|
|
478
|
+
const words = subtitle.words?.filter(isTimedEntry)
|
|
479
|
+
if (words && words.length > 0) {
|
|
480
|
+
fields.alignment = {
|
|
481
|
+
unit: 'word',
|
|
482
|
+
texts: words.map((word) => word.text ?? ''),
|
|
483
|
+
startSeconds: words.map((word) => word.start_time / 1000),
|
|
484
|
+
endSeconds: words.map((word) => word.end_time / 1000),
|
|
485
|
+
}
|
|
486
|
+
}
|
|
487
|
+
const sentences = subtitle.sentences?.filter(isTimedEntry)
|
|
488
|
+
if (sentences && sentences.length > 0) {
|
|
489
|
+
fields.segments = sentences.map((sentence) => ({
|
|
490
|
+
startSeconds: sentence.start_time / 1000,
|
|
491
|
+
endSeconds: sentence.end_time / 1000,
|
|
492
|
+
...(sentence.text !== undefined && { text: sentence.text }),
|
|
493
|
+
}))
|
|
494
|
+
}
|
|
495
|
+
return fields
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
/** Both times present, so the entry can be placed on the timeline. */
|
|
499
|
+
function isTimedEntry(
|
|
500
|
+
entry: BytePlusTTSSubtitleEntry,
|
|
501
|
+
): entry is BytePlusTTSSubtitleEntry & {
|
|
502
|
+
start_time: number
|
|
503
|
+
end_time: number
|
|
504
|
+
} {
|
|
505
|
+
return (
|
|
506
|
+
typeof entry.start_time === 'number' && typeof entry.end_time === 'number'
|
|
507
|
+
)
|
|
508
|
+
}
|
|
509
|
+
|
|
391
510
|
/**
|
|
392
511
|
* Coerce a `duration` / `original_duration` field to a usable number of
|
|
393
512
|
* seconds.
|