@tanstack/ai-byteplus 0.3.5 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,5 @@
1
1
  import { BaseTTSAdapter } from '@tanstack/ai/adapters';
2
- import { TTSOptions } from '@tanstack/ai';
2
+ import { TTSCapabilities, TTSOptions, TTSTurn } from '@tanstack/ai';
3
3
  import { InternalLogger } from '@tanstack/ai/adapter-internals';
4
4
  import { BytePlusVoiceConfig } from '../utils/client.js';
5
5
  import { BytePlusTTSModel } from '../model-meta.js';
@@ -19,6 +19,12 @@ export declare const BYTEPLUS_DEFAULT_TTS_SPEAKER = "en_female_stokie_uranus_big
19
19
  * when `speech_rate` slows it down.
20
20
  */
21
21
  export declare const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
22
+ /**
23
+ * Hard cap on the `references` array, and therefore on the number of distinct
24
+ * voices one dialogue request can use. The markers that cite them in
25
+ * `text_prompt` run `@Audio1`..`@Audio3`.
26
+ */
27
+ export declare const BYTEPLUS_TTS_MAX_REFERENCES = 3;
22
28
  /**
23
29
  * BytePlus Seed Speech text-to-speech adapter.
24
30
  *
@@ -52,6 +58,12 @@ export declare const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
52
58
  */
53
59
  export declare class BytePlusTTSAdapter<TModel extends BytePlusTTSModel = BytePlusTTSModel> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {
54
60
  readonly name: "byteplus";
61
+ /**
62
+ * Seed Audio 1.0 does multi-role dialogue in one pass, addressed through
63
+ * `references` (capped at 3 entries), and returns word/sentence timings
64
+ * when `audio_config.enable_subtitle` is set.
65
+ */
66
+ readonly capabilities: TTSCapabilities;
55
67
  private readonly apiKey;
56
68
  private readonly baseURL;
57
69
  private readonly defaultHeaders;
@@ -70,10 +82,14 @@ export declare class BytePlusTTSAdapter<TModel extends BytePlusTTSModel = BytePl
70
82
  export declare function buildTTSRequestBody(options: {
71
83
  model: string;
72
84
  text: string;
85
+ /** Dialogue turns, when the caller asked for multi-role synthesis. */
86
+ turns?: Array<TTSTurn>;
73
87
  voice: string | undefined;
74
88
  format: TTSOptions['format'] | undefined;
75
89
  speed: number | undefined;
76
90
  modelOptions: BytePlusTTSProviderOptions | undefined;
91
+ /** Core `timestamps: true` — turns on `audio_config.enable_subtitle`. */
92
+ timestamps?: boolean;
77
93
  logger: InternalLogger;
78
94
  }): {
79
95
  body: BytePlusTTSCreateRequest;
@@ -51,6 +51,12 @@ var BYTEPLUS_DEFAULT_TTS_SPEAKER = "en_female_stokie_uranus_bigtts";
51
51
  */
52
52
  var BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
53
53
  /**
54
+ * Hard cap on the `references` array, and therefore on the number of distinct
55
+ * voices one dialogue request can use. The markers that cite them in
56
+ * `text_prompt` run `@Audio1`..`@Audio3`.
57
+ */
58
+ var BYTEPLUS_TTS_MAX_REFERENCES = 3;
59
+ /**
54
60
  * BytePlus Seed Speech text-to-speech adapter.
55
61
  *
56
62
  * Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.
@@ -83,6 +89,15 @@ var BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
83
89
  */
84
90
  var BytePlusTTSAdapter = class extends BaseTTSAdapter {
85
91
  name = "byteplus";
92
+ /**
93
+ * Seed Audio 1.0 does multi-role dialogue in one pass, addressed through
94
+ * `references` (capped at 3 entries), and returns word/sentence timings
95
+ * when `audio_config.enable_subtitle` is set.
96
+ */
97
+ capabilities = {
98
+ maxSpeakers: 3,
99
+ timestamps: true
100
+ };
86
101
  apiKey;
87
102
  baseURL;
88
103
  defaultHeaders;
@@ -96,7 +111,7 @@ var BytePlusTTSAdapter = class extends BaseTTSAdapter {
96
111
  this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis);
97
112
  }
98
113
  async generateSpeech(options) {
99
- const { logger, model, text, voice, format, speed, modelOptions } = options;
114
+ const { logger, model, text, voice, format, speed, modelOptions, turns } = options;
100
115
  logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {
101
116
  provider: "byteplus",
102
117
  model
@@ -104,10 +119,12 @@ var BytePlusTTSAdapter = class extends BaseTTSAdapter {
104
119
  const { body, audioFormat, sampleRate } = buildTTSRequestBody({
105
120
  model,
106
121
  text,
122
+ turns,
107
123
  voice,
108
124
  format,
109
125
  speed,
110
126
  modelOptions,
127
+ timestamps: options.timestamps,
111
128
  logger
112
129
  });
113
130
  try {
@@ -135,6 +152,7 @@ var BytePlusTTSAdapter = class extends BaseTTSAdapter {
135
152
  ...duration !== void 0 && { duration },
136
153
  ...originalDuration !== void 0 && { originalDuration },
137
154
  ...data.subtitle !== void 0 && { subtitle: data.subtitle },
155
+ ...toAlignmentFields(data.subtitle),
138
156
  ...data.url !== void 0 && { url: data.url }
139
157
  };
140
158
  } catch (error) {
@@ -155,7 +173,7 @@ var BytePlusTTSAdapter = class extends BaseTTSAdapter {
155
173
  * `contentType`.
156
174
  */
157
175
  function buildTTSRequestBody(options) {
158
- const { model, text, voice, format, speed, modelOptions, logger } = options;
176
+ const { model, text, turns, voice, format, speed, modelOptions, logger } = options;
159
177
  const audioFormat = pickAudioFormat(modelOptions?.format, format, logger);
160
178
  const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE;
161
179
  const audioConfig = {
@@ -164,16 +182,18 @@ function buildTTSRequestBody(options) {
164
182
  };
165
183
  if (modelOptions?.pitch_rate !== void 0) audioConfig.pitch_rate = modelOptions.pitch_rate;
166
184
  if (modelOptions?.loudness_rate !== void 0) audioConfig.loudness_rate = modelOptions.loudness_rate;
167
- if (modelOptions?.enable_subtitle !== void 0) audioConfig.enable_subtitle = modelOptions.enable_subtitle;
185
+ const enableSubtitle = modelOptions?.enable_subtitle ?? (options.timestamps || void 0);
186
+ if (enableSubtitle !== void 0) audioConfig.enable_subtitle = enableSubtitle;
168
187
  const speechRate = modelOptions?.speech_rate ?? (speed !== void 0 ? toSpeechRate(speed, logger) : void 0);
169
188
  if (speechRate !== void 0) audioConfig.speech_rate = speechRate;
189
+ const dialogue = turns ? buildDialoguePrompt(turns) : void 0;
170
190
  const body = {
171
191
  model,
172
- [TTS_TEXT_FIELD]: text,
173
- references: modelOptions?.references ?? [{ speaker: modelOptions?.speaker ?? voice ?? "en_female_stokie_uranus_bigtts" }],
192
+ [TTS_TEXT_FIELD]: dialogue?.textPrompt ?? text,
193
+ references: modelOptions?.references ?? dialogue?.references ?? [{ speaker: modelOptions?.speaker ?? voice ?? "en_female_stokie_uranus_bigtts" }],
174
194
  audio_config: audioConfig
175
195
  };
176
- if (modelOptions?.watermark !== void 0) body.watermark = modelOptions.watermark;
196
+ if (modelOptions?.watermark !== void 0) body.watermark = typeof modelOptions.watermark === "boolean" ? { aigc_watermark: modelOptions.watermark } : modelOptions.watermark;
177
197
  return {
178
198
  body,
179
199
  audioFormat,
@@ -256,6 +276,59 @@ function getContentType(format, sampleRate) {
256
276
  }
257
277
  }
258
278
  /**
279
+ * Turn dialogue turns into the one-prompt form Seed Audio 1.0 takes: each
280
+ * distinct voice becomes a `references` entry, and every line of the script
281
+ * cites its voice by the reference's 1-based position (`@Audio1`..`@Audio3`),
282
+ * which is the marker convention the reference array already uses.
283
+ *
284
+ * **The prompt form is unprobed.** The flat `references[]` member shape is
285
+ * confirmed (see `BytePlusTTSReference`), and the model card documents
286
+ * multi-role dialogue and positional reference markers, but no worked
287
+ * multi-speaker example was available and no Seed Speech key existed to
288
+ * confirm how lines should cite their voice.
289
+ */
290
+ function buildDialoguePrompt(turns) {
291
+ const voices = [...new Set(turns.map((turn) => turn.voice))];
292
+ return {
293
+ textPrompt: turns.map((turn) => `@Audio${voices.indexOf(turn.voice) + 1}: ${turn.text}`).join("\n"),
294
+ references: voices.map((speaker) => ({ speaker }))
295
+ };
296
+ }
297
+ /**
298
+ * Map the BytePlus `subtitle` block onto the cross-provider `alignment` /
299
+ * `segments` fields.
300
+ *
301
+ * `words` is the finest granularity Seed Speech reports, so it becomes
302
+ * `alignment`; `sentences` is utterance segmentation, so it becomes
303
+ * `segments`. **Subtitle times are milliseconds** while the rest of the
304
+ * response is seconds — the conversion happens here, once.
305
+ *
306
+ * The raw block stays on {@link BytePlusTTSResult.subtitle} for callers
307
+ * already reading it.
308
+ */
309
+ function toAlignmentFields(subtitle) {
310
+ if (!subtitle) return {};
311
+ const fields = {};
312
+ const words = subtitle.words?.filter(isTimedEntry);
313
+ if (words && words.length > 0) fields.alignment = {
314
+ unit: "word",
315
+ texts: words.map((word) => word.text ?? ""),
316
+ startSeconds: words.map((word) => word.start_time / 1e3),
317
+ endSeconds: words.map((word) => word.end_time / 1e3)
318
+ };
319
+ const sentences = subtitle.sentences?.filter(isTimedEntry);
320
+ if (sentences && sentences.length > 0) fields.segments = sentences.map((sentence) => ({
321
+ startSeconds: sentence.start_time / 1e3,
322
+ endSeconds: sentence.end_time / 1e3,
323
+ ...sentence.text !== void 0 && { text: sentence.text }
324
+ }));
325
+ return fields;
326
+ }
327
+ /** Both times present, so the entry can be placed on the timeline. */
328
+ function isTimedEntry(entry) {
329
+ return typeof entry.start_time === "number" && typeof entry.end_time === "number";
330
+ }
331
+ /**
259
332
  * Coerce a `duration` / `original_duration` field to a usable number of
260
333
  * seconds.
261
334
  *
@@ -302,6 +375,6 @@ function byteplusSpeech(model, config) {
302
375
  return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config);
303
376
  }
304
377
  //#endregion
305
- export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BytePlusTTSAdapter, buildTTSRequestBody, byteplusSpeech, createBytePlusSpeech, getContentType, toDurationSeconds, toSpeechRate };
378
+ export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MAX_REFERENCES, BytePlusTTSAdapter, buildTTSRequestBody, byteplusSpeech, createBytePlusSpeech, getContentType, toDurationSeconds, toSpeechRate };
306
379
 
307
380
  //# sourceMappingURL=tts.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"tts.js","names":[],"sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport { generateId } from '@tanstack/ai-utils'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport {\n BYTEPLUS_VOICE_BASE_URL,\n bytePlusVoiceError,\n bytePlusVoiceHeaders,\n getBytePlusVoiceApiKeyFromEnv,\n readJsonBody,\n withBytePlusVoiceDefaults,\n} from '../utils/client'\nimport type { TTSOptions } from '@tanstack/ai'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { BytePlusVoiceConfig } from '../utils/client'\nimport type { BytePlusTTSModel } from '../model-meta'\nimport type {\n BytePlusTTSAudioConfig,\n BytePlusTTSAudioFormat,\n BytePlusTTSCreateRequest,\n BytePlusTTSCreateResponse,\n} from '../audio/wire-types'\nimport type {\n BytePlusTTSProviderOptions,\n BytePlusTTSResult,\n} from '../audio/tts-provider-options'\n\n/** Path of the synchronous Seed Speech synthesis endpoint. */\nconst TTS_CREATE_PATH = '/api/v3/tts/create'\n\n/**\n * Name of the request field carrying the text to speak.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** The endpoint\n * schema (docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01) lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`; there is no request-side `text`. The `text` spelling belongs\n * to the *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits\n * under `req_params.text`. The name stays isolated here so the two spellings\n * never get conflated.\n */\nconst TTS_TEXT_FIELD = 'text_prompt' satisfies keyof BytePlusTTSCreateRequest\n\n/**\n * Sample rate used when the caller doesn't pick one. The endpoint documents a\n * default of 40000, which is not among the rates it accepts — so the adapter\n * never relies on the server default and always sends this instead.\n */\nconst DEFAULT_SAMPLE_RATE = 24000\n\n/**\n * True when a Seed Speech envelope's `code` means success.\n *\n * Seed Speech uses `0` for success and a flat numeric code otherwise\n * (`45000010 Invalid X-Api-Key`). The success envelope has not been confirmed\n * against a live key, so `code` is accepted as a number, its string form, or\n * absent — an envelope that omits `code` entirely is treated as success, which\n * is what the HTTP status already told us.\n */\nfunction isZeroCode(code: number | string | undefined): boolean {\n if (code === undefined) return true\n return Number(code) === 0\n}\n\n/**\n * Voice used when neither `TTSOptions.voice` nor `modelOptions.speaker` is\n * set — the English female \"Stokie\" voice from the TTS 2.0 generation.\n */\nexport const BYTEPLUS_DEFAULT_TTS_SPEAKER = 'en_female_stokie_uranus_bigtts'\n\n/**\n * Hard cap on a single Seed Speech synthesis, in seconds. Longer scripts must\n * be split across calls and stitched client-side.\n *\n * The cap applies to the *pre-rate* length the service bills on — the\n * `original_duration` it returns. The delivered clip can run longer than this\n * when `speech_rate` slows it down.\n */\nexport const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120\n\n/**\n * BytePlus Seed Speech text-to-speech adapter.\n *\n * Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.\n * Two things differ from the Ark-hosted adapters in this package:\n *\n * - **Separate API key.** Seed Speech authenticates with `X-Api-Key` and its\n * own key (`BYTEPLUS_VOICE_API_KEY`). An `ARK_API_KEY` sent here is\n * rejected with `45000010 Invalid X-Api-Key`.\n * - **120 s output cap.** A single call synthesises at most\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} seconds of audio; longer scripts\n * have to be split and stitched client-side. The cap is measured before\n * `speech_rate` is applied, so a slowed clip can play for longer than that\n * — `result.duration` is the delivered length and `result.originalDuration`\n * is the metered one.\n *\n * Streaming synthesis (`tts/unidirectional` and the WebSocket API) is not\n * covered by this adapter — note that endpoint spells the text field `text`,\n * while this one uses `text_prompt`.\n *\n * @example\n * ```ts\n * const adapter = byteplusSpeech('seed-audio-1.0')\n * const result = await generateSpeech({\n * adapter,\n * text: 'welcome to the guitar store',\n * voice: 'en_female_stokie_uranus_bigtts',\n * format: 'mp3',\n * })\n * ```\n */\nexport class BytePlusTTSAdapter<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {\n readonly name = 'byteplus' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n private readonly fetchImpl: typeof fetch\n\n constructor(model: TModel, config: BytePlusVoiceConfig) {\n super(model, config)\n const resolved = withBytePlusVoiceDefaults(config)\n this.apiKey = resolved.apiKey\n this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL\n this.defaultHeaders = resolved.defaultHeaders ?? {}\n this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)\n }\n\n async generateSpeech(\n options: TTSOptions<BytePlusTTSProviderOptions>,\n ): Promise<BytePlusTTSResult> {\n const { logger, model, text, voice, format, speed, modelOptions } = options\n\n logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {\n provider: 'byteplus',\n model,\n })\n\n const { body, audioFormat, sampleRate } = buildTTSRequestBody({\n model,\n text,\n voice,\n format,\n speed,\n modelOptions,\n logger,\n })\n\n try {\n const response = await this.fetchImpl(\n `${this.baseURL}${TTS_CREATE_PATH}`,\n {\n method: 'POST',\n headers: bytePlusVoiceHeaders(this.apiKey, {\n ...this.defaultHeaders,\n // Client-generated per-request id. BytePlus echoes it in their\n // request logs, which is what support asks for when diagnosing a\n // synthesis failure.\n 'X-Api-Request-Id': newRequestId(),\n }),\n body: JSON.stringify(body),\n },\n )\n\n const payload = await readJsonBody(response)\n\n if (!response.ok) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n const data = payload as BytePlusTTSCreateResponse\n\n // Seed Speech reports status in the body, not only in the HTTP status:\n // a 200 can carry a non-zero `code`. Check it before looking at `audio`,\n // because a failed call may still return a partial or placeholder\n // payload that would otherwise be handed back as if it were valid.\n //\n // `code` is accepted as a number *or* a string. The success envelope was\n // never confirmed against a live key (no voice key yet — see\n // `audio/wire-types.ts`), and `readStringField` already tolerates both\n // forms when rendering the error, so requiring a number here would let\n // `{\"code\": \"45000010\"}` through both this gate and the one below.\n if (!isZeroCode(data.code)) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n // Belt and braces for a 200 that reports success but carries nothing to\n // play. Say that the adapter rejected it, rather than reusing the\n // envelope-error phrasing — a bare \"failed (200)\" gives no hint that the\n // response was well-formed and simply empty.\n if (typeof data.audio !== 'string' || data.audio.length === 0) {\n throw new Error(\n `BytePlus Seed Speech text-to-speech returned a success response ` +\n `with no audio (model ${model}).`,\n )\n }\n\n const duration = toDurationSeconds(data.duration)\n const originalDuration = toDurationSeconds(data.original_duration)\n\n return {\n id: generateId(this.name),\n model,\n audio: data.audio,\n format: audioFormat,\n contentType: getContentType(audioFormat, sampleRate),\n ...(duration !== undefined && { duration }),\n ...(originalDuration !== undefined && { originalDuration }),\n ...(data.subtitle !== undefined && { subtitle: data.subtitle }),\n ...(data.url !== undefined && { url: data.url }),\n }\n } catch (error) {\n logger.errors('byteplus.generateSpeech fatal', {\n error: toRunErrorPayload(error, 'byteplus.generateSpeech failed'),\n source: 'byteplus.generateSpeech',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the JSON body for `POST /api/v3/tts/create`, resolving the voice,\n * output format and rate fields in one place.\n *\n * Returns the request `body`, the resolved `audioFormat` and the\n * `sampleRate`, which the caller reports on the result and turns into a\n * `contentType`.\n */\nexport function buildTTSRequestBody(options: {\n model: string\n text: string\n voice: string | undefined\n format: TTSOptions['format'] | undefined\n speed: number | undefined\n modelOptions: BytePlusTTSProviderOptions | undefined\n logger: InternalLogger\n}): {\n body: BytePlusTTSCreateRequest\n audioFormat: BytePlusTTSAudioFormat\n sampleRate: number\n} {\n const { model, text, voice, format, speed, modelOptions, logger } = options\n\n const audioFormat = pickAudioFormat(modelOptions?.format, format, logger)\n // Always explicit: the documented server default (40000) is not one of the\n // rates the endpoint accepts, so relying on it is a coin flip.\n const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE\n\n const audioConfig: BytePlusTTSAudioConfig = {\n format: audioFormat,\n sample_rate: sampleRate,\n }\n if (modelOptions?.pitch_rate !== undefined) {\n audioConfig.pitch_rate = modelOptions.pitch_rate\n }\n if (modelOptions?.loudness_rate !== undefined) {\n audioConfig.loudness_rate = modelOptions.loudness_rate\n }\n if (modelOptions?.enable_subtitle !== undefined) {\n audioConfig.enable_subtitle = modelOptions.enable_subtitle\n }\n\n // An explicit `speech_rate` always wins over the derived one — it is the\n // native unit and the only way to reach the extremes precisely.\n const speechRate =\n modelOptions?.speech_rate ??\n (speed !== undefined ? toSpeechRate(speed, logger) : undefined)\n if (speechRate !== undefined) {\n audioConfig.speech_rate = speechRate\n }\n\n const body: BytePlusTTSCreateRequest = {\n model,\n [TTS_TEXT_FIELD]: text,\n // The voice belongs inside `references`, not at the top level — a\n // top-level `speaker` is silently ignored by the server. The flat member\n // shape here is the best-supported reading of the docs; see\n // `BytePlusTTSReference` for the unresolved part and the live-probe flag.\n references: modelOptions?.references ?? [\n {\n speaker: modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,\n },\n ],\n audio_config: audioConfig,\n }\n if (modelOptions?.watermark !== undefined) {\n body.watermark = modelOptions.watermark\n }\n\n return { body, audioFormat, sampleRate }\n}\n\n/**\n * Convert the cross-provider `TTSOptions.speed` multiplier into Seed Speech's\n * `speech_rate` percentage.\n *\n * ```\n * speech_rate = clamp(round((speed - 1) * 100), -50, 100)\n * ```\n *\n * so `0.5 → -50`, `1.0 → 0`, `1.5 → 50`, `2.0 → 100`.\n *\n * The range `-50`..`100` and both multiplier anchors (`-50` = 0.5×,\n * `100` = 2×) are documented, and hold on both the `/tts/create` and\n * `/tts/unidirectional` endpoints.\n *\n * `TTSOptions.speed` spans a wider 0.25×–4× than the endpoint supports, so\n * anything outside 0.5×–2× clamps (and warns) rather than erroring.\n */\nexport function toSpeechRate(speed: number, logger?: InternalLogger): number {\n const rate = Math.round((speed - 1) * 100)\n const clamped = Math.min(100, Math.max(-50, rate))\n if (clamped !== rate) {\n logger?.warn(\n `Speed ${speed}× is outside the range BytePlus Seed Speech documents (0.5×–2×) — clamping speech_rate from ${rate} to ${clamped}.`,\n { provider: 'byteplus', requestedSpeed: speed, speechRate: clamped },\n )\n }\n return clamped\n}\n\n/**\n * Map the cross-provider `TTSOptions.format` onto a Seed Speech output\n * format. An explicit `modelOptions.format` always wins.\n *\n * Seed Speech produces `wav`, `mp3`, `pcm` and `ogg_opus`. The generic\n * `opus` maps onto `ogg_opus`; `aac` and `flac` have no equivalent and fall\n * back to `mp3` (matching how the other non-OpenAI TTS adapters in this repo\n * handle unsupported codecs). The fallback is logged so it isn't silent.\n */\nfunction pickAudioFormat(\n override: BytePlusTTSAudioFormat | undefined,\n format: TTSOptions['format'] | undefined,\n logger: InternalLogger,\n): BytePlusTTSAudioFormat {\n if (override) return override\n if (!format) return 'mp3'\n switch (format) {\n case 'mp3':\n case 'wav':\n case 'pcm':\n return format\n case 'opus':\n return 'ogg_opus'\n case 'aac':\n case 'flac':\n logger.warn(\n `BytePlus Seed Speech does not support ${format} output — falling back to mp3. Set modelOptions.format to choose between wav, mp3, pcm and ogg_opus.`,\n { provider: 'byteplus', requestedFormat: format },\n )\n return 'mp3'\n }\n}\n\n/**\n * Build a per-request id for the `X-Api-Request-Id` header. Falls back to the\n * package's own id generator on runtimes without `crypto.randomUUID`.\n */\nfunction newRequestId(): string {\n return globalThis.crypto?.randomUUID?.() ?? generateId('byteplus-tts')\n}\n\n/**\n * MIME type for a Seed Speech output format.\n *\n * `pcm` is raw little-endian 16-bit samples, so its media type has to carry\n * the sample rate (RFC 3551/3555). The adapter always resolves a rate, so one\n * is always available here.\n */\nexport function getContentType(\n format: BytePlusTTSAudioFormat,\n sampleRate?: number,\n): string {\n switch (format) {\n case 'mp3':\n return 'audio/mpeg'\n case 'wav':\n return 'audio/wav'\n case 'ogg_opus':\n return 'audio/ogg;codecs=opus'\n case 'pcm':\n return `audio/L16;rate=${sampleRate ?? 24000}`\n }\n}\n\n/**\n * Coerce a `duration` / `original_duration` field to a usable number of\n * seconds.\n *\n * Both are documented as float **seconds**, so this only parses the string\n * form and drops values that can't be a length (zero, negative, non-numeric).\n * Note that `duration` may legitimately exceed\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} — a clip synthesised at\n * `speech_rate: -50` runs at half speed, so up to 120 s of billed audio can\n * be delivered as up to 240 s of playback. Do not \"correct\" a large value\n * here; the cap applies to `original_duration`.\n *\n * The subtitle timings are the mixed-unit exception: those are milliseconds\n * and are passed through untouched on {@link BytePlusTTSResult.subtitle}.\n */\nexport function toDurationSeconds(\n raw: number | string | undefined,\n): number | undefined {\n const value = typeof raw === 'string' ? Number(raw) : raw\n if (value === undefined || !Number.isFinite(value) || value <= 0) {\n return undefined\n }\n return value\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter with an explicit API key.\n *\n * The key is the **Seed Speech** key, not the Ark key used by the chat, image\n * and video adapters.\n *\n * @example\n * ```ts\n * const adapter = createBytePlusSpeech('seed-audio-1.0', process.env.BYTEPLUS_VOICE_API_KEY!)\n * ```\n */\nexport function createBytePlusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n apiKey: string,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return new BytePlusTTSAdapter(model, { ...config, apiKey })\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter, reading the API key from\n * `BYTEPLUS_VOICE_API_KEY`.\n *\n * @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.\n */\nexport function byteplusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config)\n}\n"],"mappings":";;;;;;AA2BA,IAAM,kBAAkB;;;;;;;;;;;;AAaxB,IAAM,iBAAiB;;;;;;AAOvB,IAAM,sBAAsB;;;;;;;;;;AAW5B,SAAS,WAAW,MAA4C;CAC9D,IAAI,SAAS,KAAA,GAAW,OAAO;CAC/B,OAAO,OAAO,IAAI,MAAM;AAC1B;;;;;AAMA,IAAa,+BAA+B;;;;;;;;;AAU5C,IAAa,kCAAkC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiC/C,IAAa,qBAAb,cAEU,eAAmD;CAC3D,OAAgB;CAEhB;CACA;CACA;CACA;CAEA,YAAY,OAAe,QAA6B;EACtD,MAAM,OAAO,MAAM;EACnB,MAAM,WAAW,0BAA0B,MAAM;EACjD,KAAK,SAAS,SAAS;EACvB,KAAK,UAAU,SAAS,WAAA;EACxB,KAAK,iBAAiB,SAAS,kBAAkB,CAAC;EAClD,KAAK,YAAY,SAAS,SAAS,WAAW,MAAM,KAAK,UAAU;CACrE;CAEA,MAAM,eACJ,SAC4B;EAC5B,MAAM,EAAE,QAAQ,OAAO,MAAM,OAAO,QAAQ,OAAO,iBAAiB;EAEpE,OAAO,QAAQ,mDAAmD,SAAS;GACzE,UAAU;GACV;EACF,CAAC;EAED,MAAM,EAAE,MAAM,aAAa,eAAe,oBAAoB;GAC5D;GACA;GACA;GACA;GACA;GACA;GACA;EACF,CAAC;EAED,IAAI;GACF,MAAM,WAAW,MAAM,KAAK,UAC1B,GAAG,KAAK,UAAU,mBAClB;IACE,QAAQ;IACR,SAAS,qBAAqB,KAAK,QAAQ;KACzC,GAAG,KAAK;KAIR,oBAAoB,aAAa;IACnC,CAAC;IACD,MAAM,KAAK,UAAU,IAAI;GAC3B,CACF;GAEA,MAAM,UAAU,MAAM,aAAa,QAAQ;GAE3C,IAAI,CAAC,SAAS,IACZ,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAGrE,MAAM,OAAO;GAYb,IAAI,CAAC,WAAW,KAAK,IAAI,GACvB,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAOrE,IAAI,OAAO,KAAK,UAAU,YAAY,KAAK,MAAM,WAAW,GAC1D,MAAM,IAAI,MACR,wFAC0B,MAAM,GAClC;GAGF,MAAM,WAAW,kBAAkB,KAAK,QAAQ;GAChD,MAAM,mBAAmB,kBAAkB,KAAK,iBAAiB;GAEjE,OAAO;IACL,IAAI,WAAW,KAAK,IAAI;IACxB;IACA,OAAO,KAAK;IACZ,QAAQ;IACR,aAAa,eAAe,aAAa,UAAU;IACnD,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;IACzC,GAAI,qBAAqB,KAAA,KAAa,EAAE,iBAAiB;IACzD,GAAI,KAAK,aAAa,KAAA,KAAa,EAAE,UAAU,KAAK,SAAS;IAC7D,GAAI,KAAK,QAAQ,KAAA,KAAa,EAAE,KAAK,KAAK,IAAI;GAChD;EACF,SAAS,OAAO;GACd,OAAO,OAAO,iCAAiC;IAC7C,OAAO,kBAAkB,OAAO,gCAAgC;IAChE,QAAQ;GACV,CAAC;GACD,MAAM;EACR;CACF;AACF;;;;;;;;;AAUA,SAAgB,oBAAoB,SAYlC;CACA,MAAM,EAAE,OAAO,MAAM,OAAO,QAAQ,OAAO,cAAc,WAAW;CAEpE,MAAM,cAAc,gBAAgB,cAAc,QAAQ,QAAQ,MAAM;CAGxE,MAAM,aAAa,cAAc,eAAe;CAEhD,MAAM,cAAsC;EAC1C,QAAQ;EACR,aAAa;CACf;CACA,IAAI,cAAc,eAAe,KAAA,GAC/B,YAAY,aAAa,aAAa;CAExC,IAAI,cAAc,kBAAkB,KAAA,GAClC,YAAY,gBAAgB,aAAa;CAE3C,IAAI,cAAc,oBAAoB,KAAA,GACpC,YAAY,kBAAkB,aAAa;CAK7C,MAAM,aACJ,cAAc,gBACb,UAAU,KAAA,IAAY,aAAa,OAAO,MAAM,IAAI,KAAA;CACvD,IAAI,eAAe,KAAA,GACjB,YAAY,cAAc;CAG5B,MAAM,OAAiC;EACrC;GACC,iBAAiB;EAKlB,YAAY,cAAc,cAAc,CACtC,EACE,SAAS,cAAc,WAAW,SAAA,iCACpC,CACF;EACA,cAAc;CAChB;CACA,IAAI,cAAc,cAAc,KAAA,GAC9B,KAAK,YAAY,aAAa;CAGhC,OAAO;EAAE;EAAM;EAAa;CAAW;AACzC;;;;;;;;;;;;;;;;;;AAmBA,SAAgB,aAAa,OAAe,QAAiC;CAC3E,MAAM,OAAO,KAAK,OAAO,QAAQ,KAAK,GAAG;CACzC,MAAM,UAAU,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC;CACjD,IAAI,YAAY,MACd,QAAQ,KACN,SAAS,MAAM,8FAA8F,KAAK,MAAM,QAAQ,IAChI;EAAE,UAAU;EAAY,gBAAgB;EAAO,YAAY;CAAQ,CACrE;CAEF,OAAO;AACT;;;;;;;;;;AAWA,SAAS,gBACP,UACA,QACA,QACwB;CACxB,IAAI,UAAU,OAAO;CACrB,IAAI,CAAC,QAAQ,OAAO;CACpB,QAAQ,QAAR;EACE,KAAK;EACL,KAAK;EACL,KAAK,OACH,OAAO;EACT,KAAK,QACH,OAAO;EACT,KAAK;EACL,KAAK;GACH,OAAO,KACL,yCAAyC,OAAO,uGAChD;IAAE,UAAU;IAAY,iBAAiB;GAAO,CAClD;GACA,OAAO;CACX;AACF;;;;;AAMA,SAAS,eAAuB;CAC9B,OAAO,WAAW,QAAQ,aAAa,KAAK,WAAW,cAAc;AACvE;;;;;;;;AASA,SAAgB,eACd,QACA,YACQ;CACR,QAAQ,QAAR;EACE,KAAK,OACH,OAAO;EACT,KAAK,OACH,OAAO;EACT,KAAK,YACH,OAAO;EACT,KAAK,OACH,OAAO,kBAAkB,cAAc;CAC3C;AACF;;;;;;;;;;;;;;;;AAiBA,SAAgB,kBACd,KACoB;CACpB,MAAM,QAAQ,OAAO,QAAQ,WAAW,OAAO,GAAG,IAAI;CACtD,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,GAC7D;CAEF,OAAO;AACT;;;;;;;;;;;;AAaA,SAAgB,qBAGd,OACA,QACA,QAC4B;CAC5B,OAAO,IAAI,mBAAmB,OAAO;EAAE,GAAG;EAAQ;CAAO,CAAC;AAC5D;;;;;;;AAQA,SAAgB,eAGd,OACA,QAC4B;CAC5B,OAAO,qBAAqB,OAAO,8BAA8B,GAAG,MAAM;AAC5E"}
1
+ {"version":3,"file":"tts.js","names":[],"sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport { generateId } from '@tanstack/ai-utils'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport {\n BYTEPLUS_VOICE_BASE_URL,\n bytePlusVoiceError,\n bytePlusVoiceHeaders,\n getBytePlusVoiceApiKeyFromEnv,\n readJsonBody,\n withBytePlusVoiceDefaults,\n} from '../utils/client'\nimport type {\n TTSAlignment,\n TTSCapabilities,\n TTSOptions,\n TTSSegment,\n TTSTurn,\n} from '@tanstack/ai'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { BytePlusVoiceConfig } from '../utils/client'\nimport type { BytePlusTTSModel } from '../model-meta'\nimport type {\n BytePlusTTSAudioConfig,\n BytePlusTTSAudioFormat,\n BytePlusTTSCreateRequest,\n BytePlusTTSCreateResponse,\n BytePlusTTSReference,\n BytePlusTTSSubtitle,\n BytePlusTTSSubtitleEntry,\n} from '../audio/wire-types'\nimport type {\n BytePlusTTSProviderOptions,\n BytePlusTTSResult,\n} from '../audio/tts-provider-options'\n\n/** Path of the synchronous Seed Speech synthesis endpoint. */\nconst TTS_CREATE_PATH = '/api/v3/tts/create'\n\n/**\n * Name of the request field carrying the text to speak.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** The endpoint\n * schema (docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01) lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`; there is no request-side `text`. The `text` spelling belongs\n * to the *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits\n * under `req_params.text`. The name stays isolated here so the two spellings\n * never get conflated.\n */\nconst TTS_TEXT_FIELD = 'text_prompt' satisfies keyof BytePlusTTSCreateRequest\n\n/**\n * Sample rate used when the caller doesn't pick one. The endpoint documents a\n * default of 40000, which is not among the rates it accepts — so the adapter\n * never relies on the server default and always sends this instead.\n */\nconst DEFAULT_SAMPLE_RATE = 24000\n\n/**\n * True when a Seed Speech envelope's `code` means success.\n *\n * Seed Speech uses `0` for success and a flat numeric code otherwise\n * (`45000010 Invalid X-Api-Key`). The success envelope has not been confirmed\n * against a live key, so `code` is accepted as a number, its string form, or\n * absent — an envelope that omits `code` entirely is treated as success, which\n * is what the HTTP status already told us.\n */\nfunction isZeroCode(code: number | string | undefined): boolean {\n if (code === undefined) return true\n return Number(code) === 0\n}\n\n/**\n * Voice used when neither `TTSOptions.voice` nor `modelOptions.speaker` is\n * set — the English female \"Stokie\" voice from the TTS 2.0 generation.\n */\nexport const BYTEPLUS_DEFAULT_TTS_SPEAKER = 'en_female_stokie_uranus_bigtts'\n\n/**\n * Hard cap on a single Seed Speech synthesis, in seconds. Longer scripts must\n * be split across calls and stitched client-side.\n *\n * The cap applies to the *pre-rate* length the service bills on — the\n * `original_duration` it returns. The delivered clip can run longer than this\n * when `speech_rate` slows it down.\n */\nexport const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120\n\n/**\n * Hard cap on the `references` array, and therefore on the number of distinct\n * voices one dialogue request can use. The markers that cite them in\n * `text_prompt` run `@Audio1`..`@Audio3`.\n */\nexport const BYTEPLUS_TTS_MAX_REFERENCES = 3\n\n/**\n * BytePlus Seed Speech text-to-speech adapter.\n *\n * Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.\n * Two things differ from the Ark-hosted adapters in this package:\n *\n * - **Separate API key.** Seed Speech authenticates with `X-Api-Key` and its\n * own key (`BYTEPLUS_VOICE_API_KEY`). An `ARK_API_KEY` sent here is\n * rejected with `45000010 Invalid X-Api-Key`.\n * - **120 s output cap.** A single call synthesises at most\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} seconds of audio; longer scripts\n * have to be split and stitched client-side. The cap is measured before\n * `speech_rate` is applied, so a slowed clip can play for longer than that\n * — `result.duration` is the delivered length and `result.originalDuration`\n * is the metered one.\n *\n * Streaming synthesis (`tts/unidirectional` and the WebSocket API) is not\n * covered by this adapter — note that endpoint spells the text field `text`,\n * while this one uses `text_prompt`.\n *\n * @example\n * ```ts\n * const adapter = byteplusSpeech('seed-audio-1.0')\n * const result = await generateSpeech({\n * adapter,\n * text: 'welcome to the guitar store',\n * voice: 'en_female_stokie_uranus_bigtts',\n * format: 'mp3',\n * })\n * ```\n */\nexport class BytePlusTTSAdapter<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {\n readonly name = 'byteplus' as const\n\n /**\n * Seed Audio 1.0 does multi-role dialogue in one pass, addressed through\n * `references` (capped at 3 entries), and returns word/sentence timings\n * when `audio_config.enable_subtitle` is set.\n */\n override readonly capabilities: TTSCapabilities = {\n maxSpeakers: BYTEPLUS_TTS_MAX_REFERENCES,\n timestamps: true,\n }\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n private readonly fetchImpl: typeof fetch\n\n constructor(model: TModel, config: BytePlusVoiceConfig) {\n super(model, config)\n const resolved = withBytePlusVoiceDefaults(config)\n this.apiKey = resolved.apiKey\n this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL\n this.defaultHeaders = resolved.defaultHeaders ?? {}\n this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)\n }\n\n async generateSpeech(\n options: TTSOptions<BytePlusTTSProviderOptions>,\n ): Promise<BytePlusTTSResult> {\n const { logger, model, text, voice, format, speed, modelOptions, turns } =\n options\n\n logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {\n provider: 'byteplus',\n model,\n })\n\n const { body, audioFormat, sampleRate } = buildTTSRequestBody({\n model,\n text,\n turns,\n voice,\n format,\n speed,\n modelOptions,\n timestamps: options.timestamps,\n logger,\n })\n\n try {\n const response = await this.fetchImpl(\n `${this.baseURL}${TTS_CREATE_PATH}`,\n {\n method: 'POST',\n headers: bytePlusVoiceHeaders(this.apiKey, {\n ...this.defaultHeaders,\n // Client-generated per-request id. BytePlus echoes it in their\n // request logs, which is what support asks for when diagnosing a\n // synthesis failure.\n 'X-Api-Request-Id': newRequestId(),\n }),\n body: JSON.stringify(body),\n },\n )\n\n const payload = await readJsonBody(response)\n\n if (!response.ok) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n const data = payload as BytePlusTTSCreateResponse\n\n // Seed Speech reports status in the body, not only in the HTTP status:\n // a 200 can carry a non-zero `code`. Check it before looking at `audio`,\n // because a failed call may still return a partial or placeholder\n // payload that would otherwise be handed back as if it were valid.\n //\n // `code` is accepted as a number *or* a string. The success envelope was\n // never confirmed against a live key (no voice key yet — see\n // `audio/wire-types.ts`), and `readStringField` already tolerates both\n // forms when rendering the error, so requiring a number here would let\n // `{\"code\": \"45000010\"}` through both this gate and the one below.\n if (!isZeroCode(data.code)) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n // Belt and braces for a 200 that reports success but carries nothing to\n // play. Say that the adapter rejected it, rather than reusing the\n // envelope-error phrasing — a bare \"failed (200)\" gives no hint that the\n // response was well-formed and simply empty.\n if (typeof data.audio !== 'string' || data.audio.length === 0) {\n throw new Error(\n `BytePlus Seed Speech text-to-speech returned a success response ` +\n `with no audio (model ${model}).`,\n )\n }\n\n const duration = toDurationSeconds(data.duration)\n const originalDuration = toDurationSeconds(data.original_duration)\n\n return {\n id: generateId(this.name),\n model,\n audio: data.audio,\n format: audioFormat,\n contentType: getContentType(audioFormat, sampleRate),\n ...(duration !== undefined && { duration }),\n ...(originalDuration !== undefined && { originalDuration }),\n ...(data.subtitle !== undefined && { subtitle: data.subtitle }),\n ...toAlignmentFields(data.subtitle),\n ...(data.url !== undefined && { url: data.url }),\n }\n } catch (error) {\n logger.errors('byteplus.generateSpeech fatal', {\n error: toRunErrorPayload(error, 'byteplus.generateSpeech failed'),\n source: 'byteplus.generateSpeech',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the JSON body for `POST /api/v3/tts/create`, resolving the voice,\n * output format and rate fields in one place.\n *\n * Returns the request `body`, the resolved `audioFormat` and the\n * `sampleRate`, which the caller reports on the result and turns into a\n * `contentType`.\n */\nexport function buildTTSRequestBody(options: {\n model: string\n text: string\n /** Dialogue turns, when the caller asked for multi-role synthesis. */\n turns?: Array<TTSTurn>\n voice: string | undefined\n format: TTSOptions['format'] | undefined\n speed: number | undefined\n modelOptions: BytePlusTTSProviderOptions | undefined\n /** Core `timestamps: true` — turns on `audio_config.enable_subtitle`. */\n timestamps?: boolean\n logger: InternalLogger\n}): {\n body: BytePlusTTSCreateRequest\n audioFormat: BytePlusTTSAudioFormat\n sampleRate: number\n} {\n const { model, text, turns, voice, format, speed, modelOptions, logger } =\n options\n\n const audioFormat = pickAudioFormat(modelOptions?.format, format, logger)\n // Always explicit: the documented server default (40000) is not one of the\n // rates the endpoint accepts, so relying on it is a coin flip.\n const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE\n\n const audioConfig: BytePlusTTSAudioConfig = {\n format: audioFormat,\n sample_rate: sampleRate,\n }\n if (modelOptions?.pitch_rate !== undefined) {\n audioConfig.pitch_rate = modelOptions.pitch_rate\n }\n if (modelOptions?.loudness_rate !== undefined) {\n audioConfig.loudness_rate = modelOptions.loudness_rate\n }\n // `modelOptions.enable_subtitle` still wins, so an explicit `false` can\n // opt out of the flag even when core asked for timestamps. A `timestamps`\n // the caller never set stays off the body entirely.\n const enableSubtitle =\n modelOptions?.enable_subtitle ?? (options.timestamps || undefined)\n if (enableSubtitle !== undefined) {\n audioConfig.enable_subtitle = enableSubtitle\n }\n\n // An explicit `speech_rate` always wins over the derived one — it is the\n // native unit and the only way to reach the extremes precisely.\n const speechRate =\n modelOptions?.speech_rate ??\n (speed !== undefined ? toSpeechRate(speed, logger) : undefined)\n if (speechRate !== undefined) {\n audioConfig.speech_rate = speechRate\n }\n\n const dialogue = turns ? buildDialoguePrompt(turns) : undefined\n\n const body: BytePlusTTSCreateRequest = {\n model,\n [TTS_TEXT_FIELD]: dialogue?.textPrompt ?? text,\n // The voice belongs inside `references`, not at the top level — a\n // top-level `speaker` is silently ignored by the server.\n references: modelOptions?.references ??\n dialogue?.references ?? [\n {\n speaker:\n modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,\n },\n ],\n audio_config: audioConfig,\n }\n if (modelOptions?.watermark !== undefined) {\n // The endpoint wants an object. `true` means the audible marker, which is\n // what a caller passing a boolean is asking for.\n body.watermark =\n typeof modelOptions.watermark === 'boolean'\n ? { aigc_watermark: modelOptions.watermark }\n : modelOptions.watermark\n }\n\n return { body, audioFormat, sampleRate }\n}\n\n/**\n * Convert the cross-provider `TTSOptions.speed` multiplier into Seed Speech's\n * `speech_rate` percentage.\n *\n * ```\n * speech_rate = clamp(round((speed - 1) * 100), -50, 100)\n * ```\n *\n * so `0.5 → -50`, `1.0 → 0`, `1.5 → 50`, `2.0 → 100`.\n *\n * The range `-50`..`100` and both multiplier anchors (`-50` = 0.5×,\n * `100` = 2×) are documented, and hold on both the `/tts/create` and\n * `/tts/unidirectional` endpoints.\n *\n * `TTSOptions.speed` spans a wider 0.25×–4× than the endpoint supports, so\n * anything outside 0.5×–2× clamps (and warns) rather than erroring.\n */\nexport function toSpeechRate(speed: number, logger?: InternalLogger): number {\n const rate = Math.round((speed - 1) * 100)\n const clamped = Math.min(100, Math.max(-50, rate))\n if (clamped !== rate) {\n logger?.warn(\n `Speed ${speed}× is outside the range BytePlus Seed Speech documents (0.5×–2×) — clamping speech_rate from ${rate} to ${clamped}.`,\n { provider: 'byteplus', requestedSpeed: speed, speechRate: clamped },\n )\n }\n return clamped\n}\n\n/**\n * Map the cross-provider `TTSOptions.format` onto a Seed Speech output\n * format. An explicit `modelOptions.format` always wins.\n *\n * Seed Speech produces `wav`, `mp3`, `pcm` and `ogg_opus`. The generic\n * `opus` maps onto `ogg_opus`; `aac` and `flac` have no equivalent and fall\n * back to `mp3` (matching how the other non-OpenAI TTS adapters in this repo\n * handle unsupported codecs). The fallback is logged so it isn't silent.\n */\nfunction pickAudioFormat(\n override: BytePlusTTSAudioFormat | undefined,\n format: TTSOptions['format'] | undefined,\n logger: InternalLogger,\n): BytePlusTTSAudioFormat {\n if (override) return override\n if (!format) return 'mp3'\n switch (format) {\n case 'mp3':\n case 'wav':\n case 'pcm':\n return format\n case 'opus':\n return 'ogg_opus'\n case 'aac':\n case 'flac':\n logger.warn(\n `BytePlus Seed Speech does not support ${format} output — falling back to mp3. Set modelOptions.format to choose between wav, mp3, pcm and ogg_opus.`,\n { provider: 'byteplus', requestedFormat: format },\n )\n return 'mp3'\n }\n}\n\n/**\n * Build a per-request id for the `X-Api-Request-Id` header. Falls back to the\n * package's own id generator on runtimes without `crypto.randomUUID`.\n */\nfunction newRequestId(): string {\n return globalThis.crypto?.randomUUID?.() ?? generateId('byteplus-tts')\n}\n\n/**\n * MIME type for a Seed Speech output format.\n *\n * `pcm` is raw little-endian 16-bit samples, so its media type has to carry\n * the sample rate (RFC 3551/3555). The adapter always resolves a rate, so one\n * is always available here.\n */\nexport function getContentType(\n format: BytePlusTTSAudioFormat,\n sampleRate?: number,\n): string {\n switch (format) {\n case 'mp3':\n return 'audio/mpeg'\n case 'wav':\n return 'audio/wav'\n case 'ogg_opus':\n return 'audio/ogg;codecs=opus'\n case 'pcm':\n return `audio/L16;rate=${sampleRate ?? 24000}`\n }\n}\n\n/**\n * Turn dialogue turns into the one-prompt form Seed Audio 1.0 takes: each\n * distinct voice becomes a `references` entry, and every line of the script\n * cites its voice by the reference's 1-based position (`@Audio1`..`@Audio3`),\n * which is the marker convention the reference array already uses.\n *\n * **The prompt form is unprobed.** The flat `references[]` member shape is\n * confirmed (see `BytePlusTTSReference`), and the model card documents\n * multi-role dialogue and positional reference markers, but no worked\n * multi-speaker example was available and no Seed Speech key existed to\n * confirm how lines should cite their voice.\n */\nfunction buildDialoguePrompt(turns: Array<TTSTurn>): {\n textPrompt: string\n references: Array<BytePlusTTSReference>\n} {\n const voices = [...new Set(turns.map((turn) => turn.voice))]\n return {\n textPrompt: turns\n .map((turn) => `@Audio${voices.indexOf(turn.voice) + 1}: ${turn.text}`)\n .join('\\n'),\n references: voices.map((speaker) => ({ speaker })),\n }\n}\n\n/**\n * Map the BytePlus `subtitle` block onto the cross-provider `alignment` /\n * `segments` fields.\n *\n * `words` is the finest granularity Seed Speech reports, so it becomes\n * `alignment`; `sentences` is utterance segmentation, so it becomes\n * `segments`. **Subtitle times are milliseconds** while the rest of the\n * response is seconds — the conversion happens here, once.\n *\n * The raw block stays on {@link BytePlusTTSResult.subtitle} for callers\n * already reading it.\n */\nfunction toAlignmentFields(subtitle: BytePlusTTSSubtitle | undefined): {\n alignment?: TTSAlignment\n segments?: Array<TTSSegment>\n} {\n if (!subtitle) return {}\n const fields: { alignment?: TTSAlignment; segments?: Array<TTSSegment> } = {}\n const words = subtitle.words?.filter(isTimedEntry)\n if (words && words.length > 0) {\n fields.alignment = {\n unit: 'word',\n texts: words.map((word) => word.text ?? ''),\n startSeconds: words.map((word) => word.start_time / 1000),\n endSeconds: words.map((word) => word.end_time / 1000),\n }\n }\n const sentences = subtitle.sentences?.filter(isTimedEntry)\n if (sentences && sentences.length > 0) {\n fields.segments = sentences.map((sentence) => ({\n startSeconds: sentence.start_time / 1000,\n endSeconds: sentence.end_time / 1000,\n ...(sentence.text !== undefined && { text: sentence.text }),\n }))\n }\n return fields\n}\n\n/** Both times present, so the entry can be placed on the timeline. */\nfunction isTimedEntry(\n entry: BytePlusTTSSubtitleEntry,\n): entry is BytePlusTTSSubtitleEntry & {\n start_time: number\n end_time: number\n} {\n return (\n typeof entry.start_time === 'number' && typeof entry.end_time === 'number'\n )\n}\n\n/**\n * Coerce a `duration` / `original_duration` field to a usable number of\n * seconds.\n *\n * Both are documented as float **seconds**, so this only parses the string\n * form and drops values that can't be a length (zero, negative, non-numeric).\n * Note that `duration` may legitimately exceed\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} — a clip synthesised at\n * `speech_rate: -50` runs at half speed, so up to 120 s of billed audio can\n * be delivered as up to 240 s of playback. Do not \"correct\" a large value\n * here; the cap applies to `original_duration`.\n *\n * The subtitle timings are the mixed-unit exception: those are milliseconds\n * and are passed through untouched on {@link BytePlusTTSResult.subtitle}.\n */\nexport function toDurationSeconds(\n raw: number | string | undefined,\n): number | undefined {\n const value = typeof raw === 'string' ? Number(raw) : raw\n if (value === undefined || !Number.isFinite(value) || value <= 0) {\n return undefined\n }\n return value\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter with an explicit API key.\n *\n * The key is the **Seed Speech** key, not the Ark key used by the chat, image\n * and video adapters.\n *\n * @example\n * ```ts\n * const adapter = createBytePlusSpeech('seed-audio-1.0', process.env.BYTEPLUS_VOICE_API_KEY!)\n * ```\n */\nexport function createBytePlusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n apiKey: string,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return new BytePlusTTSAdapter(model, { ...config, apiKey })\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter, reading the API key from\n * `BYTEPLUS_VOICE_API_KEY`.\n *\n * @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.\n */\nexport function byteplusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config)\n}\n"],"mappings":";;;;;;AAoCA,IAAM,kBAAkB;;;;;;;;;;;;AAaxB,IAAM,iBAAiB;;;;;;AAOvB,IAAM,sBAAsB;;;;;;;;;;AAW5B,SAAS,WAAW,MAA4C;CAC9D,IAAI,SAAS,KAAA,GAAW,OAAO;CAC/B,OAAO,OAAO,IAAI,MAAM;AAC1B;;;;;AAMA,IAAa,+BAA+B;;;;;;;;;AAU5C,IAAa,kCAAkC;;;;;;AAO/C,IAAa,8BAA8B;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiC3C,IAAa,qBAAb,cAEU,eAAmD;CAC3D,OAAgB;;;;;;CAOhB,eAAkD;EAChD,aAAA;EACA,YAAY;CACd;CAEA;CACA;CACA;CACA;CAEA,YAAY,OAAe,QAA6B;EACtD,MAAM,OAAO,MAAM;EACnB,MAAM,WAAW,0BAA0B,MAAM;EACjD,KAAK,SAAS,SAAS;EACvB,KAAK,UAAU,SAAS,WAAA;EACxB,KAAK,iBAAiB,SAAS,kBAAkB,CAAC;EAClD,KAAK,YAAY,SAAS,SAAS,WAAW,MAAM,KAAK,UAAU;CACrE;CAEA,MAAM,eACJ,SAC4B;EAC5B,MAAM,EAAE,QAAQ,OAAO,MAAM,OAAO,QAAQ,OAAO,cAAc,UAC/D;EAEF,OAAO,QAAQ,mDAAmD,SAAS;GACzE,UAAU;GACV;EACF,CAAC;EAED,MAAM,EAAE,MAAM,aAAa,eAAe,oBAAoB;GAC5D;GACA;GACA;GACA;GACA;GACA;GACA;GACA,YAAY,QAAQ;GACpB;EACF,CAAC;EAED,IAAI;GACF,MAAM,WAAW,MAAM,KAAK,UAC1B,GAAG,KAAK,UAAU,mBAClB;IACE,QAAQ;IACR,SAAS,qBAAqB,KAAK,QAAQ;KACzC,GAAG,KAAK;KAIR,oBAAoB,aAAa;IACnC,CAAC;IACD,MAAM,KAAK,UAAU,IAAI;GAC3B,CACF;GAEA,MAAM,UAAU,MAAM,aAAa,QAAQ;GAE3C,IAAI,CAAC,SAAS,IACZ,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAGrE,MAAM,OAAO;GAYb,IAAI,CAAC,WAAW,KAAK,IAAI,GACvB,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAOrE,IAAI,OAAO,KAAK,UAAU,YAAY,KAAK,MAAM,WAAW,GAC1D,MAAM,IAAI,MACR,wFAC0B,MAAM,GAClC;GAGF,MAAM,WAAW,kBAAkB,KAAK,QAAQ;GAChD,MAAM,mBAAmB,kBAAkB,KAAK,iBAAiB;GAEjE,OAAO;IACL,IAAI,WAAW,KAAK,IAAI;IACxB;IACA,OAAO,KAAK;IACZ,QAAQ;IACR,aAAa,eAAe,aAAa,UAAU;IACnD,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;IACzC,GAAI,qBAAqB,KAAA,KAAa,EAAE,iBAAiB;IACzD,GAAI,KAAK,aAAa,KAAA,KAAa,EAAE,UAAU,KAAK,SAAS;IAC7D,GAAG,kBAAkB,KAAK,QAAQ;IAClC,GAAI,KAAK,QAAQ,KAAA,KAAa,EAAE,KAAK,KAAK,IAAI;GAChD;EACF,SAAS,OAAO;GACd,OAAO,OAAO,iCAAiC;IAC7C,OAAO,kBAAkB,OAAO,gCAAgC;IAChE,QAAQ;GACV,CAAC;GACD,MAAM;EACR;CACF;AACF;;;;;;;;;AAUA,SAAgB,oBAAoB,SAgBlC;CACA,MAAM,EAAE,OAAO,MAAM,OAAO,OAAO,QAAQ,OAAO,cAAc,WAC9D;CAEF,MAAM,cAAc,gBAAgB,cAAc,QAAQ,QAAQ,MAAM;CAGxE,MAAM,aAAa,cAAc,eAAe;CAEhD,MAAM,cAAsC;EAC1C,QAAQ;EACR,aAAa;CACf;CACA,IAAI,cAAc,eAAe,KAAA,GAC/B,YAAY,aAAa,aAAa;CAExC,IAAI,cAAc,kBAAkB,KAAA,GAClC,YAAY,gBAAgB,aAAa;CAK3C,MAAM,iBACJ,cAAc,oBAAoB,QAAQ,cAAc,KAAA;CAC1D,IAAI,mBAAmB,KAAA,GACrB,YAAY,kBAAkB;CAKhC,MAAM,aACJ,cAAc,gBACb,UAAU,KAAA,IAAY,aAAa,OAAO,MAAM,IAAI,KAAA;CACvD,IAAI,eAAe,KAAA,GACjB,YAAY,cAAc;CAG5B,MAAM,WAAW,QAAQ,oBAAoB,KAAK,IAAI,KAAA;CAEtD,MAAM,OAAiC;EACrC;GACC,iBAAiB,UAAU,cAAc;EAG1C,YAAY,cAAc,cACxB,UAAU,cAAc,CACtB,EACE,SACE,cAAc,WAAW,SAAA,iCAC7B,CACF;EACF,cAAc;CAChB;CACA,IAAI,cAAc,cAAc,KAAA,GAG9B,KAAK,YACH,OAAO,aAAa,cAAc,YAC9B,EAAE,gBAAgB,aAAa,UAAU,IACzC,aAAa;CAGrB,OAAO;EAAE;EAAM;EAAa;CAAW;AACzC;;;;;;;;;;;;;;;;;;AAmBA,SAAgB,aAAa,OAAe,QAAiC;CAC3E,MAAM,OAAO,KAAK,OAAO,QAAQ,KAAK,GAAG;CACzC,MAAM,UAAU,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC;CACjD,IAAI,YAAY,MACd,QAAQ,KACN,SAAS,MAAM,8FAA8F,KAAK,MAAM,QAAQ,IAChI;EAAE,UAAU;EAAY,gBAAgB;EAAO,YAAY;CAAQ,CACrE;CAEF,OAAO;AACT;;;;;;;;;;AAWA,SAAS,gBACP,UACA,QACA,QACwB;CACxB,IAAI,UAAU,OAAO;CACrB,IAAI,CAAC,QAAQ,OAAO;CACpB,QAAQ,QAAR;EACE,KAAK;EACL,KAAK;EACL,KAAK,OACH,OAAO;EACT,KAAK,QACH,OAAO;EACT,KAAK;EACL,KAAK;GACH,OAAO,KACL,yCAAyC,OAAO,uGAChD;IAAE,UAAU;IAAY,iBAAiB;GAAO,CAClD;GACA,OAAO;CACX;AACF;;;;;AAMA,SAAS,eAAuB;CAC9B,OAAO,WAAW,QAAQ,aAAa,KAAK,WAAW,cAAc;AACvE;;;;;;;;AASA,SAAgB,eACd,QACA,YACQ;CACR,QAAQ,QAAR;EACE,KAAK,OACH,OAAO;EACT,KAAK,OACH,OAAO;EACT,KAAK,YACH,OAAO;EACT,KAAK,OACH,OAAO,kBAAkB,cAAc;CAC3C;AACF;;;;;;;;;;;;;AAcA,SAAS,oBAAoB,OAG3B;CACA,MAAM,SAAS,CAAC,GAAG,IAAI,IAAI,MAAM,KAAK,SAAS,KAAK,KAAK,CAAC,CAAC;CAC3D,OAAO;EACL,YAAY,MACT,KAAK,SAAS,SAAS,OAAO,QAAQ,KAAK,KAAK,IAAI,EAAE,IAAI,KAAK,MAAM,CAAC,CACtE,KAAK,IAAI;EACZ,YAAY,OAAO,KAAK,aAAa,EAAE,QAAQ,EAAE;CACnD;AACF;;;;;;;;;;;;;AAcA,SAAS,kBAAkB,UAGzB;CACA,IAAI,CAAC,UAAU,OAAO,CAAC;CACvB,MAAM,SAAqE,CAAC;CAC5E,MAAM,QAAQ,SAAS,OAAO,OAAO,YAAY;CACjD,IAAI,SAAS,MAAM,SAAS,GAC1B,OAAO,YAAY;EACjB,MAAM;EACN,OAAO,MAAM,KAAK,SAAS,KAAK,QAAQ,EAAE;EAC1C,cAAc,MAAM,KAAK,SAAS,KAAK,aAAa,GAAI;EACxD,YAAY,MAAM,KAAK,SAAS,KAAK,WAAW,GAAI;CACtD;CAEF,MAAM,YAAY,SAAS,WAAW,OAAO,YAAY;CACzD,IAAI,aAAa,UAAU,SAAS,GAClC,OAAO,WAAW,UAAU,KAAK,cAAc;EAC7C,cAAc,SAAS,aAAa;EACpC,YAAY,SAAS,WAAW;EAChC,GAAI,SAAS,SAAS,KAAA,KAAa,EAAE,MAAM,SAAS,KAAK;CAC3D,EAAE;CAEJ,OAAO;AACT;;AAGA,SAAS,aACP,OAIA;CACA,OACE,OAAO,MAAM,eAAe,YAAY,OAAO,MAAM,aAAa;AAEtE;;;;;;;;;;;;;;;;AAiBA,SAAgB,kBACd,KACoB;CACpB,MAAM,QAAQ,OAAO,QAAQ,WAAW,OAAO,GAAG,IAAI;CACtD,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,GAC7D;CAEF,OAAO;AACT;;;;;;;;;;;;AAaA,SAAgB,qBAGd,OACA,QACA,QAC4B;CAC5B,OAAO,IAAI,mBAAmB,OAAO;EAAE,GAAG;EAAQ;CAAO,CAAC;AAC5D;;;;;;;AAQA,SAAgB,eAGd,OACA,QAC4B;CAC5B,OAAO,qBAAqB,OAAO,8BAA8B,GAAG,MAAM;AAC5E"}
@@ -1,4 +1,4 @@
1
- import { BytePlusTTSAudioFormat, BytePlusTTSReference, BytePlusTTSSampleRate, BytePlusTTSSubtitle } from './wire-types.js';
1
+ import { BytePlusTTSAudioFormat, BytePlusTTSReference, BytePlusTTSSampleRate, BytePlusTTSSubtitle, BytePlusTTSWatermark } from './wire-types.js';
2
2
  import { TTSResult } from '@tanstack/ai';
3
3
  /**
4
4
  * Seed Speech voice identifier (`speaker` on the wire).
@@ -72,10 +72,13 @@ export interface BytePlusTTSProviderOptions {
72
72
  */
73
73
  enable_subtitle?: boolean;
74
74
  /**
75
- * Watermark the generated audio. The field name is confirmed against the
76
- * endpoint schema; the boolean type is assumed and unprobed.
75
+ * Watermark the generated audio.
76
+ *
77
+ * Seed Audio takes an object with two independent markers — an audible
78
+ * `aigc_watermark` appended to the clip and an `aigc_metadata` block written
79
+ * into the audio header. `true` is shorthand for `{ aigc_watermark: true }`.
77
80
  */
78
- watermark?: boolean;
81
+ watermark?: boolean | BytePlusTTSWatermark;
79
82
  }
80
83
  /**
81
84
  * BytePlus-specific extension of `TTSResult`.
@@ -46,15 +46,16 @@ export type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number];
46
46
  * audio references (30 s / 10 MB each) and 1 image reference (10 MB), and
47
47
  * image references are mutually exclusive with audio ones.
48
48
  *
49
- * **Member object shape is unresolved — must live-probe when the Seed Speech
50
- * key lands.** The docs list the member fields flat (`speaker | audio_data |
51
- * audio_url | image_data | image_url`) without a worked example, so whether
52
- * the server wants a flat `{ speaker }` or a discriminated `{ type, ... }`
53
- * could not be settled. The adapter sends the flat reading — see
54
- * `buildTTSRequestBody` in `../adapters/tts`.
49
+ * The member fields are flat (`speaker | audio_data | audio_url | image_data |
50
+ * image_url`), confirmed against
51
+ * docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01.
55
52
  */
56
53
  export interface BytePlusTTSReference {
57
- /** Stock voice id, e.g. `en_female_stokie_uranus_bigtts`. */
54
+ /**
55
+ * Voice id, e.g. `en_female_stokie_uranus_bigtts`. Either a stock TTS 2.0
56
+ * voice or the id of a voice you cloned through Voice Replication — this
57
+ * field is the join between replication and synthesis.
58
+ */
58
59
  speaker?: string;
59
60
  /** URL of a reference clip to clone (≤30 s, ≤10 MB). */
60
61
  audio_url?: string;
@@ -88,6 +89,36 @@ export interface BytePlusTTSAudioConfig {
88
89
  /** Emit sentence and word timings in the response. Defaults to `false`. */
89
90
  enable_subtitle?: boolean;
90
91
  }
92
+ /**
93
+ * `watermark` block of a TTS request.
94
+ *
95
+ * Two independent markers, both off by default:
96
+ *
97
+ * - `aigc_watermark` — **explicit**: appends an audible rhythm marker to the
98
+ * end of the clip.
99
+ * - `aigc_metadata` — **implicit**: writes provenance metadata into the audio
100
+ * header. Nothing is written unless `enable` is `true`.
101
+ *
102
+ * This is an object, not a boolean — unlike Seedream images and Seedance
103
+ * video, where `watermark` genuinely is a boolean.
104
+ */
105
+ export interface BytePlusTTSWatermark {
106
+ /** Append an audible rhythm marker to the end of the audio. Default `false`. */
107
+ aigc_watermark?: boolean;
108
+ /** Provenance metadata written into the audio header. */
109
+ aigc_metadata?: {
110
+ /** Write the metadata. Default `false` — the rest is ignored without it. */
111
+ enable?: boolean;
112
+ /** Name or code of the synthesis provider. */
113
+ content_producer?: string;
114
+ /** Content production id. */
115
+ produce_id?: string;
116
+ /** Name or code of the distribution provider. */
117
+ content_propagator?: string;
118
+ /** Content distribution id. */
119
+ propagate_id?: string;
120
+ };
121
+ }
91
122
  /** Request body for `POST /api/v3/tts/create` — exactly five fields. */
92
123
  export interface BytePlusTTSCreateRequest {
93
124
  /** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */
@@ -109,11 +140,8 @@ export interface BytePlusTTSCreateRequest {
109
140
  /** Voice selection and cloning references. See {@link BytePlusTTSReference}. */
110
141
  references?: Array<BytePlusTTSReference>;
111
142
  audio_config?: BytePlusTTSAudioConfig;
112
- /**
113
- * Watermark the generated audio. The field name is confirmed; the boolean
114
- * type is assumed by analogy with Seedream's `watermark` and unprobed.
115
- */
116
- watermark?: boolean;
143
+ /** Watermark the generated audio. See {@link BytePlusTTSWatermark}. */
144
+ watermark?: BytePlusTTSWatermark;
117
145
  }
118
146
  /**
119
147
  * One timed entry of a TTS subtitle track.
@@ -1 +1 @@
1
- {"version":3,"file":"wire-types.js","names":[],"sources":["../../../src/audio/wire-types.ts"],"sourcesContent":["/**\n * Minimal wire types for the BytePlus **Seed Speech** HTTP API (TTS + ASR).\n *\n * Seed Speech is a separate product from Ark: it lives on\n * `voice.ap-southeast-1.bytepluses.com`, authenticates with `X-Api-Key`\n * (a different key from `ARK_API_KEY`), and returns a flat numeric error\n * envelope instead of Ark's OpenAI-shaped one.\n *\n * Only the fields the adapters read or write are modelled here — this is a\n * hand-written subset, not a generated schema.\n *\n * Provenance:\n * - Endpoints, auth header, format/rate ranges and the 120 s TTS output cap:\n * BytePlus Seed Speech docs (`docs.byteplus.com/en/docs/byteplusvoice`),\n * captured in the Phase 0 research notes.\n * - Error envelope `{code, message}`: verified live — an Ark key sent as\n * `X-Api-Key` returns HTTP 401 `{\"code\":45000010,\"message\":\"Invalid X-Api-Key\"}`.\n * - ASR request/response shape (`user`/`audio`/`request` in, `audio_info` +\n * `result.utterances` out, all timings in **milliseconds**): the Volcengine\n * flash-recognition reference the BytePlus endpoint is derived from\n * (`docs.volcengine.com/docs/6561/1631584`).\n *\n * No Seed Speech API key was available when these were written, so the TTS\n * response fields are documented-but-unverified; the adapters parse them\n * defensively rather than assuming they are always present.\n */\n\n// ============================================================================\n// TTS — POST /api/v3/tts/create\n// ============================================================================\n\n/** Output container/codec accepted by `audio_config.format`. */\nexport type BytePlusTTSAudioFormat = 'wav' | 'mp3' | 'pcm' | 'ogg_opus'\n\n/**\n * Sample rates `audio_config.sample_rate` accepts.\n *\n * The docs also state a *default* of 40000, which is not one of the valid\n * values — a documentation bug. The adapter therefore always sends an\n * explicit rate rather than relying on the server default.\n */\nexport const BYTEPLUS_TTS_SAMPLE_RATES = [\n 8000, 16000, 24000, 32000, 44100, 48000,\n] as const\n\n/** A sample rate `audio_config.sample_rate` accepts. */\nexport type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number]\n\n/**\n * One entry of the request's `references` array.\n *\n * This is where the voice lives: `speaker` names a stock voice, while\n * `audio_url` / `audio_data` supply reference clips to clone (the `@Audio1`..\n * `@Audio3` markers in `text_prompt` address them positionally). Exactly one\n * of `speaker`, `audio_url` or `audio_data` may be set per entry; at most 3\n * audio references (30 s / 10 MB each) and 1 image reference (10 MB), and\n * image references are mutually exclusive with audio ones.\n *\n * **Member object shape is unresolved — must live-probe when the Seed Speech\n * key lands.** The docs list the member fields flat (`speaker | audio_data |\n * audio_url | image_data | image_url`) without a worked example, so whether\n * the server wants a flat `{ speaker }` or a discriminated `{ type, ... }`\n * could not be settled. The adapter sends the flat reading — see\n * `buildTTSRequestBody` in `../adapters/tts`.\n */\nexport interface BytePlusTTSReference {\n /** Stock voice id, e.g. `en_female_stokie_uranus_bigtts`. */\n speaker?: string\n /** URL of a reference clip to clone (≤30 s, ≤10 MB). */\n audio_url?: string\n /** Base64 reference clip to clone (≤30 s, ≤10 MB). */\n audio_data?: string\n /** URL of a reference image (≤10 MB). Mutually exclusive with audio refs. */\n image_url?: string\n /** Base64 reference image (≤10 MB). Mutually exclusive with audio refs. */\n image_data?: string\n}\n\n/**\n * `audio_config` block of a TTS request — exactly six fields.\n *\n * The three `*_rate` fields are integer percentages relative to the voice's\n * neutral delivery, not multipliers.\n */\nexport interface BytePlusTTSAudioConfig {\n /** Output format. Defaults to `wav` server-side. */\n format?: BytePlusTTSAudioFormat\n /**\n * Output sample rate in Hz. See {@link BYTEPLUS_TTS_SAMPLE_RATES} for the\n * valid values and for why the adapter always sends one.\n */\n sample_rate?: number\n /** Speaking rate, `-50`..`100`. `-50` = 0.5×, `0` = 1×, `100` = 2×. */\n speech_rate?: number\n /** Loudness adjustment, `-50`..`100`. `0` is the voice's natural level. */\n loudness_rate?: number\n /** Pitch adjustment, `-12`..`12`. `0` is the voice's natural pitch. */\n pitch_rate?: number\n /** Emit sentence and word timings in the response. Defaults to `false`. */\n enable_subtitle?: boolean\n}\n\n/** Request body for `POST /api/v3/tts/create` — exactly five fields. */\nexport interface BytePlusTTSCreateRequest {\n /** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */\n model: string\n /**\n * The text to speak (≤3000 chars). Dual-purpose: either literal text or a\n * natural-language description of the delivery, and the place the\n * `@Audio1`..`@Audio3` reference markers go.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** This endpoint\n * has no request-side `text` field. The `text` spelling belongs to the\n * *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits under\n * `req_params.text`. Confirmed against\n * docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01, which lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`.\n */\n text_prompt: string\n /** Voice selection and cloning references. See {@link BytePlusTTSReference}. */\n references?: Array<BytePlusTTSReference>\n audio_config?: BytePlusTTSAudioConfig\n /**\n * Watermark the generated audio. The field name is confirmed; the boolean\n * type is assumed by analogy with Seedream's `watermark` and unprobed.\n */\n watermark?: boolean\n}\n\n/**\n * One timed entry of a TTS subtitle track.\n *\n * **Times are milliseconds** — unlike the response's `duration` fields, which\n * are seconds. The endpoint genuinely mixes units.\n */\nexport interface BytePlusTTSSubtitleEntry {\n text?: string\n start_time?: number\n end_time?: number\n}\n\n/** Sentence- and word-level timings returned when `enable_subtitle` is set. */\nexport interface BytePlusTTSSubtitle {\n sentences?: Array<BytePlusTTSSubtitleEntry>\n words?: Array<BytePlusTTSSubtitleEntry>\n}\n\n/** Response body for `POST /api/v3/tts/create`. */\nexport interface BytePlusTTSCreateResponse {\n /**\n * Status code — `0` on success, a flat error code otherwise.\n *\n * Typed as `number | string` because only the *error* envelope was verified\n * live (HTTP 401 `{\"code\":45000010,…}`); the success envelope's shape is\n * docs-derived and no voice key was available to confirm it. Treating a\n * string code as \"not a failure\" would return a failed 200 as success, so\n * the adapter coerces before comparing — see `isZeroCode` in\n * `adapters/tts.ts`.\n */\n code?: number | string\n message?: string\n /** Base64-encoded audio in the requested `audio_config.format`. */\n audio?: string\n /**\n * Length of the delivered audio in **seconds** (float), after `speech_rate`\n * is applied. This can legitimately exceed 120 when the clip is slowed\n * down — the 120 s cap applies to {@link BytePlusTTSCreateResponse.original_duration}.\n */\n duration?: number | string\n /**\n * Length in **seconds** (float) before rate adjustment. This is the billing\n * basis and is capped at 120.\n */\n original_duration?: number | string\n /** Temporary download URL for the same audio. Expires after ~2 hours. */\n url?: string\n /** Sentence and word timings, present when `enable_subtitle` was set. */\n subtitle?: BytePlusTTSSubtitle\n}\n\n// ============================================================================\n// ASR — POST /api/v3/auc/bigmodel/recognize/flash\n// ============================================================================\n\n/**\n * Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo\n * model. The flash endpoint takes no `model` field in its body — the model is\n * chosen entirely by this header.\n */\nexport const BYTEPLUS_ASR_RESOURCE_ID = 'volc.seedasr.auc_turbo'\n\n/** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */\nexport const BYTEPLUS_ASR_RESOURCE_HEADER = 'X-Api-Resource-Id'\n\n/**\n * Audio input. Exactly one of `url` or `data` is sent — the endpoint accepts\n * files up to 2 hours long / 100 MB.\n */\nexport interface BytePlusASRAudio {\n /** Publicly reachable URL of the audio file. */\n url?: string\n /** Base64-encoded audio bytes. */\n data?: string\n /** Container hint, e.g. `mp3`, `wav`, `ogg`. */\n format?: string\n}\n\n/** `request` block of a recognition call. */\nexport interface BytePlusASRRequestOptions {\n /** Recognition model family. Defaults to `bigmodel`. */\n model_name?: string\n /** Inverse text normalisation (spoken numbers → digits). */\n enable_itn?: boolean\n /** Insert punctuation. */\n enable_punc?: boolean\n /** Disfluency removal (\"um\", repeated words). */\n enable_ddc?: boolean\n /** Attach per-utterance speaker labels. */\n enable_speaker_info?: boolean\n /** Return the `utterances` breakdown as well as the flat transcript. */\n show_utterances?: boolean\n /** Spoken language hint, e.g. `en-US`. */\n language?: string\n}\n\n/** Request body for `POST /api/v3/auc/bigmodel/recognize/flash`. */\nexport interface BytePlusASRRecognizeRequest {\n user?: { uid?: string }\n audio: BytePlusASRAudio\n request?: BytePlusASRRequestOptions\n}\n\n/** One recognised word. `start_time` / `end_time` are milliseconds. */\nexport interface BytePlusASRWord {\n text?: string\n start_time?: number\n end_time?: number\n confidence?: number\n}\n\n/** One recognised utterance. `start_time` / `end_time` are milliseconds. */\nexport interface BytePlusASRUtterance {\n text?: string\n start_time?: number\n end_time?: number\n words?: Array<BytePlusASRWord>\n /**\n * Extra per-utterance annotations. Speaker labels arrive here when\n * `enable_speaker_info` is set; the exact key is read defensively because it\n * could not be confirmed against a live response.\n */\n additions?: Record<string, string>\n}\n\nexport interface BytePlusASRResult {\n text?: string\n utterances?: Array<BytePlusASRUtterance>\n}\n\n/**\n * Response body for `POST /api/v3/auc/bigmodel/recognize/flash`.\n *\n * The Volcengine-lineage wire shape nests everything under `result`; BytePlus'\n * prose docs describe the same payload as \"transcript + utterances\", so the\n * flat spelling is tolerated as a fallback.\n */\nexport interface BytePlusASRRecognizeResponse {\n /** `duration` is the audio length in **milliseconds**. */\n audio_info?: { duration?: number }\n result?: BytePlusASRResult\n /** Flat alias for `result.text`. */\n transcript?: string\n /** Flat alias for `result.utterances`. */\n utterances?: Array<BytePlusASRUtterance>\n}\n\n// ============================================================================\n// Errors\n// ============================================================================\n\n/**\n * Seed Speech error envelope: a flat numeric `code` plus a `message`, e.g.\n * `{\"code\": 45000010, \"message\": \"Invalid X-Api-Key\"}` (verified live on a\n * 401). Format it with `bytePlusVoiceError` from `../utils/client`.\n */\nexport interface BytePlusVoiceErrorBody {\n code?: number\n message?: string\n}\n"],"mappings":";;;;;;;;AAyCA,IAAa,4BAA4B;CACvC;CAAM;CAAO;CAAO;CAAO;CAAO;AACpC;;;;;;AAmJA,IAAa,2BAA2B;;AAGxC,IAAa,+BAA+B"}
1
+ {"version":3,"file":"wire-types.js","names":[],"sources":["../../../src/audio/wire-types.ts"],"sourcesContent":["/**\n * Minimal wire types for the BytePlus **Seed Speech** HTTP API (TTS + ASR).\n *\n * Seed Speech is a separate product from Ark: it lives on\n * `voice.ap-southeast-1.bytepluses.com`, authenticates with `X-Api-Key`\n * (a different key from `ARK_API_KEY`), and returns a flat numeric error\n * envelope instead of Ark's OpenAI-shaped one.\n *\n * Only the fields the adapters read or write are modelled here — this is a\n * hand-written subset, not a generated schema.\n *\n * Provenance:\n * - Endpoints, auth header, format/rate ranges and the 120 s TTS output cap:\n * BytePlus Seed Speech docs (`docs.byteplus.com/en/docs/byteplusvoice`),\n * captured in the Phase 0 research notes.\n * - Error envelope `{code, message}`: verified live — an Ark key sent as\n * `X-Api-Key` returns HTTP 401 `{\"code\":45000010,\"message\":\"Invalid X-Api-Key\"}`.\n * - ASR request/response shape (`user`/`audio`/`request` in, `audio_info` +\n * `result.utterances` out, all timings in **milliseconds**): the Volcengine\n * flash-recognition reference the BytePlus endpoint is derived from\n * (`docs.volcengine.com/docs/6561/1631584`).\n *\n * No Seed Speech API key was available when these were written, so the TTS\n * response fields are documented-but-unverified; the adapters parse them\n * defensively rather than assuming they are always present.\n */\n\n// ============================================================================\n// TTS — POST /api/v3/tts/create\n// ============================================================================\n\n/** Output container/codec accepted by `audio_config.format`. */\nexport type BytePlusTTSAudioFormat = 'wav' | 'mp3' | 'pcm' | 'ogg_opus'\n\n/**\n * Sample rates `audio_config.sample_rate` accepts.\n *\n * The docs also state a *default* of 40000, which is not one of the valid\n * values — a documentation bug. The adapter therefore always sends an\n * explicit rate rather than relying on the server default.\n */\nexport const BYTEPLUS_TTS_SAMPLE_RATES = [\n 8000, 16000, 24000, 32000, 44100, 48000,\n] as const\n\n/** A sample rate `audio_config.sample_rate` accepts. */\nexport type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number]\n\n/**\n * One entry of the request's `references` array.\n *\n * This is where the voice lives: `speaker` names a stock voice, while\n * `audio_url` / `audio_data` supply reference clips to clone (the `@Audio1`..\n * `@Audio3` markers in `text_prompt` address them positionally). Exactly one\n * of `speaker`, `audio_url` or `audio_data` may be set per entry; at most 3\n * audio references (30 s / 10 MB each) and 1 image reference (10 MB), and\n * image references are mutually exclusive with audio ones.\n *\n * The member fields are flat (`speaker | audio_data | audio_url | image_data |\n * image_url`), confirmed against\n * docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01.\n */\nexport interface BytePlusTTSReference {\n /**\n * Voice id, e.g. `en_female_stokie_uranus_bigtts`. Either a stock TTS 2.0\n * voice or the id of a voice you cloned through Voice Replication — this\n * field is the join between replication and synthesis.\n */\n speaker?: string\n /** URL of a reference clip to clone (≤30 s, ≤10 MB). */\n audio_url?: string\n /** Base64 reference clip to clone (≤30 s, ≤10 MB). */\n audio_data?: string\n /** URL of a reference image (≤10 MB). Mutually exclusive with audio refs. */\n image_url?: string\n /** Base64 reference image (≤10 MB). Mutually exclusive with audio refs. */\n image_data?: string\n}\n\n/**\n * `audio_config` block of a TTS request — exactly six fields.\n *\n * The three `*_rate` fields are integer percentages relative to the voice's\n * neutral delivery, not multipliers.\n */\nexport interface BytePlusTTSAudioConfig {\n /** Output format. Defaults to `wav` server-side. */\n format?: BytePlusTTSAudioFormat\n /**\n * Output sample rate in Hz. See {@link BYTEPLUS_TTS_SAMPLE_RATES} for the\n * valid values and for why the adapter always sends one.\n */\n sample_rate?: number\n /** Speaking rate, `-50`..`100`. `-50` = 0.5×, `0` = 1×, `100` = 2×. */\n speech_rate?: number\n /** Loudness adjustment, `-50`..`100`. `0` is the voice's natural level. */\n loudness_rate?: number\n /** Pitch adjustment, `-12`..`12`. `0` is the voice's natural pitch. */\n pitch_rate?: number\n /** Emit sentence and word timings in the response. Defaults to `false`. */\n enable_subtitle?: boolean\n}\n\n/**\n * `watermark` block of a TTS request.\n *\n * Two independent markers, both off by default:\n *\n * - `aigc_watermark` — **explicit**: appends an audible rhythm marker to the\n * end of the clip.\n * - `aigc_metadata` — **implicit**: writes provenance metadata into the audio\n * header. Nothing is written unless `enable` is `true`.\n *\n * This is an object, not a boolean — unlike Seedream images and Seedance\n * video, where `watermark` genuinely is a boolean.\n */\nexport interface BytePlusTTSWatermark {\n /** Append an audible rhythm marker to the end of the audio. Default `false`. */\n aigc_watermark?: boolean\n /** Provenance metadata written into the audio header. */\n aigc_metadata?: {\n /** Write the metadata. Default `false` — the rest is ignored without it. */\n enable?: boolean\n /** Name or code of the synthesis provider. */\n content_producer?: string\n /** Content production id. */\n produce_id?: string\n /** Name or code of the distribution provider. */\n content_propagator?: string\n /** Content distribution id. */\n propagate_id?: string\n }\n}\n\n/** Request body for `POST /api/v3/tts/create` — exactly five fields. */\nexport interface BytePlusTTSCreateRequest {\n /** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */\n model: string\n /**\n * The text to speak (≤3000 chars). Dual-purpose: either literal text or a\n * natural-language description of the delivery, and the place the\n * `@Audio1`..`@Audio3` reference markers go.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** This endpoint\n * has no request-side `text` field. The `text` spelling belongs to the\n * *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits under\n * `req_params.text`. Confirmed against\n * docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01, which lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`.\n */\n text_prompt: string\n /** Voice selection and cloning references. See {@link BytePlusTTSReference}. */\n references?: Array<BytePlusTTSReference>\n audio_config?: BytePlusTTSAudioConfig\n /** Watermark the generated audio. See {@link BytePlusTTSWatermark}. */\n watermark?: BytePlusTTSWatermark\n}\n\n/**\n * One timed entry of a TTS subtitle track.\n *\n * **Times are milliseconds** — unlike the response's `duration` fields, which\n * are seconds. The endpoint genuinely mixes units.\n */\nexport interface BytePlusTTSSubtitleEntry {\n text?: string\n start_time?: number\n end_time?: number\n}\n\n/** Sentence- and word-level timings returned when `enable_subtitle` is set. */\nexport interface BytePlusTTSSubtitle {\n sentences?: Array<BytePlusTTSSubtitleEntry>\n words?: Array<BytePlusTTSSubtitleEntry>\n}\n\n/** Response body for `POST /api/v3/tts/create`. */\nexport interface BytePlusTTSCreateResponse {\n /**\n * Status code — `0` on success, a flat error code otherwise.\n *\n * Typed as `number | string` because only the *error* envelope was verified\n * live (HTTP 401 `{\"code\":45000010,…}`); the success envelope's shape is\n * docs-derived and no voice key was available to confirm it. Treating a\n * string code as \"not a failure\" would return a failed 200 as success, so\n * the adapter coerces before comparing — see `isZeroCode` in\n * `adapters/tts.ts`.\n */\n code?: number | string\n message?: string\n /** Base64-encoded audio in the requested `audio_config.format`. */\n audio?: string\n /**\n * Length of the delivered audio in **seconds** (float), after `speech_rate`\n * is applied. This can legitimately exceed 120 when the clip is slowed\n * down — the 120 s cap applies to {@link BytePlusTTSCreateResponse.original_duration}.\n */\n duration?: number | string\n /**\n * Length in **seconds** (float) before rate adjustment. This is the billing\n * basis and is capped at 120.\n */\n original_duration?: number | string\n /** Temporary download URL for the same audio. Expires after ~2 hours. */\n url?: string\n /** Sentence and word timings, present when `enable_subtitle` was set. */\n subtitle?: BytePlusTTSSubtitle\n}\n\n// ============================================================================\n// ASR — POST /api/v3/auc/bigmodel/recognize/flash\n// ============================================================================\n\n/**\n * Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo\n * model. The flash endpoint takes no `model` field in its body — the model is\n * chosen entirely by this header.\n */\nexport const BYTEPLUS_ASR_RESOURCE_ID = 'volc.seedasr.auc_turbo'\n\n/** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */\nexport const BYTEPLUS_ASR_RESOURCE_HEADER = 'X-Api-Resource-Id'\n\n/**\n * Audio input. Exactly one of `url` or `data` is sent — the endpoint accepts\n * files up to 2 hours long / 100 MB.\n */\nexport interface BytePlusASRAudio {\n /** Publicly reachable URL of the audio file. */\n url?: string\n /** Base64-encoded audio bytes. */\n data?: string\n /** Container hint, e.g. `mp3`, `wav`, `ogg`. */\n format?: string\n}\n\n/** `request` block of a recognition call. */\nexport interface BytePlusASRRequestOptions {\n /** Recognition model family. Defaults to `bigmodel`. */\n model_name?: string\n /** Inverse text normalisation (spoken numbers → digits). */\n enable_itn?: boolean\n /** Insert punctuation. */\n enable_punc?: boolean\n /** Disfluency removal (\"um\", repeated words). */\n enable_ddc?: boolean\n /** Attach per-utterance speaker labels. */\n enable_speaker_info?: boolean\n /** Return the `utterances` breakdown as well as the flat transcript. */\n show_utterances?: boolean\n /** Spoken language hint, e.g. `en-US`. */\n language?: string\n}\n\n/** Request body for `POST /api/v3/auc/bigmodel/recognize/flash`. */\nexport interface BytePlusASRRecognizeRequest {\n user?: { uid?: string }\n audio: BytePlusASRAudio\n request?: BytePlusASRRequestOptions\n}\n\n/** One recognised word. `start_time` / `end_time` are milliseconds. */\nexport interface BytePlusASRWord {\n text?: string\n start_time?: number\n end_time?: number\n confidence?: number\n}\n\n/** One recognised utterance. `start_time` / `end_time` are milliseconds. */\nexport interface BytePlusASRUtterance {\n text?: string\n start_time?: number\n end_time?: number\n words?: Array<BytePlusASRWord>\n /**\n * Extra per-utterance annotations. Speaker labels arrive here when\n * `enable_speaker_info` is set; the exact key is read defensively because it\n * could not be confirmed against a live response.\n */\n additions?: Record<string, string>\n}\n\nexport interface BytePlusASRResult {\n text?: string\n utterances?: Array<BytePlusASRUtterance>\n}\n\n/**\n * Response body for `POST /api/v3/auc/bigmodel/recognize/flash`.\n *\n * The Volcengine-lineage wire shape nests everything under `result`; BytePlus'\n * prose docs describe the same payload as \"transcript + utterances\", so the\n * flat spelling is tolerated as a fallback.\n */\nexport interface BytePlusASRRecognizeResponse {\n /** `duration` is the audio length in **milliseconds**. */\n audio_info?: { duration?: number }\n result?: BytePlusASRResult\n /** Flat alias for `result.text`. */\n transcript?: string\n /** Flat alias for `result.utterances`. */\n utterances?: Array<BytePlusASRUtterance>\n}\n\n// ============================================================================\n// Errors\n// ============================================================================\n\n/**\n * Seed Speech error envelope: a flat numeric `code` plus a `message`, e.g.\n * `{\"code\": 45000010, \"message\": \"Invalid X-Api-Key\"}` (verified live on a\n * 401). Format it with `bytePlusVoiceError` from `../utils/client`.\n */\nexport interface BytePlusVoiceErrorBody {\n code?: number\n message?: string\n}\n"],"mappings":";;;;;;;;AAyCA,IAAa,4BAA4B;CACvC;CAAM;CAAO;CAAO;CAAO;CAAO;AACpC;;;;;;AAgLA,IAAa,2BAA2B;;AAGxC,IAAa,+BAA+B"}
@@ -3,13 +3,13 @@ export type { BytePlusVideoConfig } from './adapters/video.js';
3
3
  export { parseBytePlusVideoSize, resolveBytePlusVideoResolution, resolveBytePlusVideoSize, supportsAudioOnlyReference, supportsLastFrame, supportsReferenceMedia, } from './video/video-provider-options.js';
4
4
  export type { BytePlusVideoModelProviderOptionsByName, BytePlusVideoOutputFormat, BytePlusVideoProviderOptions, BytePlusVideoServiceTier, } from './video/video-provider-options.js';
5
5
  export type { BytePlusVideoContentPart, BytePlusVideoContentRole, BytePlusVideoCreateRequest, BytePlusVideoCreateResponse, BytePlusVideoTask, BytePlusVideoTaskContent, BytePlusVideoTaskError, BytePlusVideoTaskListItem, BytePlusVideoTaskListResponse, BytePlusVideoTaskStatus, BytePlusVideoTaskUsage, } from './video/wire-types.js';
6
- export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BytePlusTTSAdapter, byteplusSpeech, createBytePlusSpeech, toSpeechRate, } from './adapters/tts.js';
6
+ export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MAX_REFERENCES, BytePlusTTSAdapter, byteplusSpeech, createBytePlusSpeech, toSpeechRate, } from './adapters/tts.js';
7
7
  export type { BytePlusTTSProviderOptions, BytePlusTTSResult, BytePlusTTSVoice, } from './audio/tts-provider-options.js';
8
8
  export { BytePlusTranscriptionAdapter, byteplusTranscription, createBytePlusTranscription, } from './adapters/transcription.js';
9
9
  export type { BytePlusTranscriptionWord } from './adapters/transcription.js';
10
10
  export type { BytePlusTranscriptionProviderOptions } from './audio/transcription-provider-options.js';
11
11
  export { BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_TTS_SAMPLE_RATES, } from './audio/wire-types.js';
12
- export type { BytePlusASRAudio, BytePlusASRRecognizeRequest, BytePlusASRRecognizeResponse, BytePlusASRResult, BytePlusASRUtterance, BytePlusASRWord, BytePlusTTSAudioConfig, BytePlusTTSAudioFormat, BytePlusTTSCreateRequest, BytePlusTTSCreateResponse, BytePlusTTSReference, BytePlusTTSSampleRate, BytePlusTTSSubtitle, BytePlusTTSSubtitleEntry, BytePlusVoiceErrorBody, } from './audio/wire-types.js';
12
+ export type { BytePlusASRAudio, BytePlusASRRecognizeRequest, BytePlusASRRecognizeResponse, BytePlusASRResult, BytePlusASRUtterance, BytePlusASRWord, BytePlusTTSAudioConfig, BytePlusTTSAudioFormat, BytePlusTTSCreateRequest, BytePlusTTSCreateResponse, BytePlusTTSReference, BytePlusTTSSampleRate, BytePlusTTSSubtitle, BytePlusTTSSubtitleEntry, BytePlusTTSWatermark, BytePlusVoiceErrorBody, } from './audio/wire-types.js';
13
13
  export { BytePlusImageAdapter, byteplusImage, createBytePlusImage, } from './adapters/image.js';
14
14
  export type { BytePlusImageConfig } from './adapters/image.js';
15
15
  export { BYTEPLUS_IMAGE_MAX_PROMPT_WORDS, BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES, BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS, parseBytePlusImageSize, } from './image/image-provider-options.js';
package/dist/esm/index.js CHANGED
@@ -2,10 +2,10 @@ import { BYTEPLUS_ARK_BASE_URL, BYTEPLUS_VOICE_BASE_URL, bytePlusArkError, byteP
2
2
  import { BYTEPLUS_CHAT_MODELS, BYTEPLUS_IMAGE_MAX_REFERENCE_IMAGES, BYTEPLUS_IMAGE_MODELS, BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS, BYTEPLUS_THINKING_SUMMARY_MODELS, BYTEPLUS_TRANSCRIPTION_MODELS, BYTEPLUS_TTS_MODELS, BYTEPLUS_VIDEO_DURATIONS, BYTEPLUS_VIDEO_FALLBACK_DURATIONS, BYTEPLUS_VIDEO_MODELS, emitsEncryptedContent, getBytePlusVideoDurationOptions, isKnownBytePlusVideoModel, supportsStructuredOutput } from "./model-meta.js";
3
3
  import { parseBytePlusVideoSize, resolveBytePlusVideoResolution, resolveBytePlusVideoSize, supportsAudioOnlyReference, supportsLastFrame, supportsReferenceMedia } from "./video/video-provider-options.js";
4
4
  import { BytePlusVideoAdapter, byteplusVideo, createBytePlusVideo } from "./adapters/video.js";
5
- import { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BytePlusTTSAdapter, byteplusSpeech, createBytePlusSpeech, toSpeechRate } from "./adapters/tts.js";
5
+ import { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MAX_REFERENCES, BytePlusTTSAdapter, byteplusSpeech, createBytePlusSpeech, toSpeechRate } from "./adapters/tts.js";
6
6
  import { BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_TTS_SAMPLE_RATES } from "./audio/wire-types.js";
7
7
  import { BytePlusTranscriptionAdapter, byteplusTranscription, createBytePlusTranscription } from "./adapters/transcription.js";
8
8
  import { BYTEPLUS_IMAGE_MAX_PROMPT_WORDS, BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES, BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS, parseBytePlusImageSize } from "./image/image-provider-options.js";
9
9
  import { BytePlusImageAdapter, byteplusImage, createBytePlusImage } from "./adapters/image.js";
10
10
  import { BytePlusTextAdapter, byteplusText, createBytePlusText } from "./adapters/text.js";
11
- export { BYTEPLUS_ARK_BASE_URL, BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_CHAT_MODELS, BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_IMAGE_MAX_PROMPT_WORDS, BYTEPLUS_IMAGE_MAX_REFERENCE_IMAGES, BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES, BYTEPLUS_IMAGE_MODELS, BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS, BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS, BYTEPLUS_THINKING_SUMMARY_MODELS, BYTEPLUS_TRANSCRIPTION_MODELS, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MODELS, BYTEPLUS_TTS_SAMPLE_RATES, BYTEPLUS_VIDEO_DURATIONS, BYTEPLUS_VIDEO_FALLBACK_DURATIONS, BYTEPLUS_VIDEO_MODELS, BYTEPLUS_VOICE_BASE_URL, BytePlusImageAdapter, BytePlusTTSAdapter, BytePlusTextAdapter, BytePlusTranscriptionAdapter, BytePlusVideoAdapter, bytePlusArkError, bytePlusArkHeaders, bytePlusVoiceError, bytePlusVoiceHeaders, byteplusImage, byteplusSpeech, byteplusText, byteplusTranscription, byteplusVideo, createBytePlusImage, createBytePlusSpeech, createBytePlusText, createBytePlusTranscription, createBytePlusVideo, emitsEncryptedContent, getBytePlusArkApiKeyFromEnv, getBytePlusVideoDurationOptions, getBytePlusVoiceApiKeyFromEnv, isKnownBytePlusVideoModel, parseBytePlusImageSize, parseBytePlusVideoSize, resolveBytePlusVideoResolution, resolveBytePlusVideoSize, supportsAudioOnlyReference, supportsLastFrame, supportsReferenceMedia, supportsStructuredOutput, toSpeechRate, withBytePlusArkDefaults, withBytePlusVoiceDefaults };
11
+ export { BYTEPLUS_ARK_BASE_URL, BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_CHAT_MODELS, BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_IMAGE_MAX_PROMPT_WORDS, BYTEPLUS_IMAGE_MAX_REFERENCE_IMAGES, BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES, BYTEPLUS_IMAGE_MODELS, BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS, BYTEPLUS_STRUCTURED_OUTPUT_CHAT_MODELS, BYTEPLUS_THINKING_SUMMARY_MODELS, BYTEPLUS_TRANSCRIPTION_MODELS, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BYTEPLUS_TTS_MAX_REFERENCES, BYTEPLUS_TTS_MODELS, BYTEPLUS_TTS_SAMPLE_RATES, BYTEPLUS_VIDEO_DURATIONS, BYTEPLUS_VIDEO_FALLBACK_DURATIONS, BYTEPLUS_VIDEO_MODELS, BYTEPLUS_VOICE_BASE_URL, BytePlusImageAdapter, BytePlusTTSAdapter, BytePlusTextAdapter, BytePlusTranscriptionAdapter, BytePlusVideoAdapter, bytePlusArkError, bytePlusArkHeaders, bytePlusVoiceError, bytePlusVoiceHeaders, byteplusImage, byteplusSpeech, byteplusText, byteplusTranscription, byteplusVideo, createBytePlusImage, createBytePlusSpeech, createBytePlusText, createBytePlusTranscription, createBytePlusVideo, emitsEncryptedContent, getBytePlusArkApiKeyFromEnv, getBytePlusVideoDurationOptions, getBytePlusVoiceApiKeyFromEnv, isKnownBytePlusVideoModel, parseBytePlusImageSize, parseBytePlusVideoSize, resolveBytePlusVideoResolution, resolveBytePlusVideoSize, supportsAudioOnlyReference, supportsLastFrame, supportsReferenceMedia, supportsStructuredOutput, toSpeechRate, withBytePlusArkDefaults, withBytePlusVoiceDefaults };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai-byteplus",
3
- "version": "0.3.5",
3
+ "version": "0.4.1",
4
4
  "description": "BytePlus ModelArk adapter for TanStack AI: Seed LLM chat, Seedance video, Seedream image, and Seed Speech TTS/ASR.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -54,15 +54,15 @@
54
54
  "devDependencies": {
55
55
  "@vitest/coverage-v8": "4.1.10",
56
56
  "vite": "^8.2.1",
57
- "@tanstack/ai": "0.54.0"
57
+ "@tanstack/ai": "0.57.0"
58
58
  },
59
59
  "peerDependencies": {
60
- "@tanstack/ai": "^0.54.0"
60
+ "@tanstack/ai": "^0.57.0"
61
61
  },
62
62
  "dependencies": {
63
63
  "openai": "^6.41.0",
64
64
  "@tanstack/ai-utils": "^0.4.0",
65
- "@tanstack/openai-base": "^0.10.11"
65
+ "@tanstack/openai-base": "^0.10.14"
66
66
  },
67
67
  "scripts": {
68
68
  "build": "vite build",
@@ -9,7 +9,13 @@ import {
9
9
  readJsonBody,
10
10
  withBytePlusVoiceDefaults,
11
11
  } from '../utils/client'
12
- import type { TTSOptions } from '@tanstack/ai'
12
+ import type {
13
+ TTSAlignment,
14
+ TTSCapabilities,
15
+ TTSOptions,
16
+ TTSSegment,
17
+ TTSTurn,
18
+ } from '@tanstack/ai'
13
19
  import type { InternalLogger } from '@tanstack/ai/adapter-internals'
14
20
  import type { BytePlusVoiceConfig } from '../utils/client'
15
21
  import type { BytePlusTTSModel } from '../model-meta'
@@ -18,6 +24,9 @@ import type {
18
24
  BytePlusTTSAudioFormat,
19
25
  BytePlusTTSCreateRequest,
20
26
  BytePlusTTSCreateResponse,
27
+ BytePlusTTSReference,
28
+ BytePlusTTSSubtitle,
29
+ BytePlusTTSSubtitleEntry,
21
30
  } from '../audio/wire-types'
22
31
  import type {
23
32
  BytePlusTTSProviderOptions,
@@ -77,6 +86,13 @@ export const BYTEPLUS_DEFAULT_TTS_SPEAKER = 'en_female_stokie_uranus_bigtts'
77
86
  */
78
87
  export const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120
79
88
 
89
+ /**
90
+ * Hard cap on the `references` array, and therefore on the number of distinct
91
+ * voices one dialogue request can use. The markers that cite them in
92
+ * `text_prompt` run `@Audio1`..`@Audio3`.
93
+ */
94
+ export const BYTEPLUS_TTS_MAX_REFERENCES = 3
95
+
80
96
  /**
81
97
  * BytePlus Seed Speech text-to-speech adapter.
82
98
  *
@@ -113,6 +129,16 @@ export class BytePlusTTSAdapter<
113
129
  > extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {
114
130
  readonly name = 'byteplus' as const
115
131
 
132
+ /**
133
+ * Seed Audio 1.0 does multi-role dialogue in one pass, addressed through
134
+ * `references` (capped at 3 entries), and returns word/sentence timings
135
+ * when `audio_config.enable_subtitle` is set.
136
+ */
137
+ override readonly capabilities: TTSCapabilities = {
138
+ maxSpeakers: BYTEPLUS_TTS_MAX_REFERENCES,
139
+ timestamps: true,
140
+ }
141
+
116
142
  private readonly apiKey: string
117
143
  private readonly baseURL: string
118
144
  private readonly defaultHeaders: Record<string, string>
@@ -130,7 +156,8 @@ export class BytePlusTTSAdapter<
130
156
  async generateSpeech(
131
157
  options: TTSOptions<BytePlusTTSProviderOptions>,
132
158
  ): Promise<BytePlusTTSResult> {
133
- const { logger, model, text, voice, format, speed, modelOptions } = options
159
+ const { logger, model, text, voice, format, speed, modelOptions, turns } =
160
+ options
134
161
 
135
162
  logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {
136
163
  provider: 'byteplus',
@@ -140,10 +167,12 @@ export class BytePlusTTSAdapter<
140
167
  const { body, audioFormat, sampleRate } = buildTTSRequestBody({
141
168
  model,
142
169
  text,
170
+ turns,
143
171
  voice,
144
172
  format,
145
173
  speed,
146
174
  modelOptions,
175
+ timestamps: options.timestamps,
147
176
  logger,
148
177
  })
149
178
 
@@ -208,6 +237,7 @@ export class BytePlusTTSAdapter<
208
237
  ...(duration !== undefined && { duration }),
209
238
  ...(originalDuration !== undefined && { originalDuration }),
210
239
  ...(data.subtitle !== undefined && { subtitle: data.subtitle }),
240
+ ...toAlignmentFields(data.subtitle),
211
241
  ...(data.url !== undefined && { url: data.url }),
212
242
  }
213
243
  } catch (error) {
@@ -231,17 +261,22 @@ export class BytePlusTTSAdapter<
231
261
  export function buildTTSRequestBody(options: {
232
262
  model: string
233
263
  text: string
264
+ /** Dialogue turns, when the caller asked for multi-role synthesis. */
265
+ turns?: Array<TTSTurn>
234
266
  voice: string | undefined
235
267
  format: TTSOptions['format'] | undefined
236
268
  speed: number | undefined
237
269
  modelOptions: BytePlusTTSProviderOptions | undefined
270
+ /** Core `timestamps: true` — turns on `audio_config.enable_subtitle`. */
271
+ timestamps?: boolean
238
272
  logger: InternalLogger
239
273
  }): {
240
274
  body: BytePlusTTSCreateRequest
241
275
  audioFormat: BytePlusTTSAudioFormat
242
276
  sampleRate: number
243
277
  } {
244
- const { model, text, voice, format, speed, modelOptions, logger } = options
278
+ const { model, text, turns, voice, format, speed, modelOptions, logger } =
279
+ options
245
280
 
246
281
  const audioFormat = pickAudioFormat(modelOptions?.format, format, logger)
247
282
  // Always explicit: the documented server default (40000) is not one of the
@@ -258,8 +293,13 @@ export function buildTTSRequestBody(options: {
258
293
  if (modelOptions?.loudness_rate !== undefined) {
259
294
  audioConfig.loudness_rate = modelOptions.loudness_rate
260
295
  }
261
- if (modelOptions?.enable_subtitle !== undefined) {
262
- audioConfig.enable_subtitle = modelOptions.enable_subtitle
296
+ // `modelOptions.enable_subtitle` still wins, so an explicit `false` can
297
+ // opt out of the flag even when core asked for timestamps. A `timestamps`
298
+ // the caller never set stays off the body entirely.
299
+ const enableSubtitle =
300
+ modelOptions?.enable_subtitle ?? (options.timestamps || undefined)
301
+ if (enableSubtitle !== undefined) {
302
+ audioConfig.enable_subtitle = enableSubtitle
263
303
  }
264
304
 
265
305
  // An explicit `speech_rate` always wins over the derived one — it is the
@@ -271,22 +311,29 @@ export function buildTTSRequestBody(options: {
271
311
  audioConfig.speech_rate = speechRate
272
312
  }
273
313
 
314
+ const dialogue = turns ? buildDialoguePrompt(turns) : undefined
315
+
274
316
  const body: BytePlusTTSCreateRequest = {
275
317
  model,
276
- [TTS_TEXT_FIELD]: text,
318
+ [TTS_TEXT_FIELD]: dialogue?.textPrompt ?? text,
277
319
  // The voice belongs inside `references`, not at the top level — a
278
- // top-level `speaker` is silently ignored by the server. The flat member
279
- // shape here is the best-supported reading of the docs; see
280
- // `BytePlusTTSReference` for the unresolved part and the live-probe flag.
281
- references: modelOptions?.references ?? [
282
- {
283
- speaker: modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,
284
- },
285
- ],
320
+ // top-level `speaker` is silently ignored by the server.
321
+ references: modelOptions?.references ??
322
+ dialogue?.references ?? [
323
+ {
324
+ speaker:
325
+ modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,
326
+ },
327
+ ],
286
328
  audio_config: audioConfig,
287
329
  }
288
330
  if (modelOptions?.watermark !== undefined) {
289
- body.watermark = modelOptions.watermark
331
+ // The endpoint wants an object. `true` means the audible marker, which is
332
+ // what a caller passing a boolean is asking for.
333
+ body.watermark =
334
+ typeof modelOptions.watermark === 'boolean'
335
+ ? { aigc_watermark: modelOptions.watermark }
336
+ : modelOptions.watermark
290
337
  }
291
338
 
292
339
  return { body, audioFormat, sampleRate }
@@ -385,6 +432,81 @@ export function getContentType(
385
432
  }
386
433
  }
387
434
 
435
+ /**
436
+ * Turn dialogue turns into the one-prompt form Seed Audio 1.0 takes: each
437
+ * distinct voice becomes a `references` entry, and every line of the script
438
+ * cites its voice by the reference's 1-based position (`@Audio1`..`@Audio3`),
439
+ * which is the marker convention the reference array already uses.
440
+ *
441
+ * **The prompt form is unprobed.** The flat `references[]` member shape is
442
+ * confirmed (see `BytePlusTTSReference`), and the model card documents
443
+ * multi-role dialogue and positional reference markers, but no worked
444
+ * multi-speaker example was available and no Seed Speech key existed to
445
+ * confirm how lines should cite their voice.
446
+ */
447
+ function buildDialoguePrompt(turns: Array<TTSTurn>): {
448
+ textPrompt: string
449
+ references: Array<BytePlusTTSReference>
450
+ } {
451
+ const voices = [...new Set(turns.map((turn) => turn.voice))]
452
+ return {
453
+ textPrompt: turns
454
+ .map((turn) => `@Audio${voices.indexOf(turn.voice) + 1}: ${turn.text}`)
455
+ .join('\n'),
456
+ references: voices.map((speaker) => ({ speaker })),
457
+ }
458
+ }
459
+
460
+ /**
461
+ * Map the BytePlus `subtitle` block onto the cross-provider `alignment` /
462
+ * `segments` fields.
463
+ *
464
+ * `words` is the finest granularity Seed Speech reports, so it becomes
465
+ * `alignment`; `sentences` is utterance segmentation, so it becomes
466
+ * `segments`. **Subtitle times are milliseconds** while the rest of the
467
+ * response is seconds — the conversion happens here, once.
468
+ *
469
+ * The raw block stays on {@link BytePlusTTSResult.subtitle} for callers
470
+ * already reading it.
471
+ */
472
+ function toAlignmentFields(subtitle: BytePlusTTSSubtitle | undefined): {
473
+ alignment?: TTSAlignment
474
+ segments?: Array<TTSSegment>
475
+ } {
476
+ if (!subtitle) return {}
477
+ const fields: { alignment?: TTSAlignment; segments?: Array<TTSSegment> } = {}
478
+ const words = subtitle.words?.filter(isTimedEntry)
479
+ if (words && words.length > 0) {
480
+ fields.alignment = {
481
+ unit: 'word',
482
+ texts: words.map((word) => word.text ?? ''),
483
+ startSeconds: words.map((word) => word.start_time / 1000),
484
+ endSeconds: words.map((word) => word.end_time / 1000),
485
+ }
486
+ }
487
+ const sentences = subtitle.sentences?.filter(isTimedEntry)
488
+ if (sentences && sentences.length > 0) {
489
+ fields.segments = sentences.map((sentence) => ({
490
+ startSeconds: sentence.start_time / 1000,
491
+ endSeconds: sentence.end_time / 1000,
492
+ ...(sentence.text !== undefined && { text: sentence.text }),
493
+ }))
494
+ }
495
+ return fields
496
+ }
497
+
498
+ /** Both times present, so the entry can be placed on the timeline. */
499
+ function isTimedEntry(
500
+ entry: BytePlusTTSSubtitleEntry,
501
+ ): entry is BytePlusTTSSubtitleEntry & {
502
+ start_time: number
503
+ end_time: number
504
+ } {
505
+ return (
506
+ typeof entry.start_time === 'number' && typeof entry.end_time === 'number'
507
+ )
508
+ }
509
+
388
510
  /**
389
511
  * Coerce a `duration` / `original_duration` field to a usable number of
390
512
  * seconds.
@@ -3,6 +3,7 @@ import type {
3
3
  BytePlusTTSReference,
4
4
  BytePlusTTSSampleRate,
5
5
  BytePlusTTSSubtitle,
6
+ BytePlusTTSWatermark,
6
7
  } from './wire-types'
7
8
  import type { TTSResult } from '@tanstack/ai'
8
9
 
@@ -79,10 +80,13 @@ export interface BytePlusTTSProviderOptions {
79
80
  */
80
81
  enable_subtitle?: boolean
81
82
  /**
82
- * Watermark the generated audio. The field name is confirmed against the
83
- * endpoint schema; the boolean type is assumed and unprobed.
83
+ * Watermark the generated audio.
84
+ *
85
+ * Seed Audio takes an object with two independent markers — an audible
86
+ * `aigc_watermark` appended to the clip and an `aigc_metadata` block written
87
+ * into the audio header. `true` is shorthand for `{ aigc_watermark: true }`.
84
88
  */
85
- watermark?: boolean
89
+ watermark?: boolean | BytePlusTTSWatermark
86
90
  }
87
91
 
88
92
  /**
@@ -56,15 +56,16 @@ export type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number]
56
56
  * audio references (30 s / 10 MB each) and 1 image reference (10 MB), and
57
57
  * image references are mutually exclusive with audio ones.
58
58
  *
59
- * **Member object shape is unresolved — must live-probe when the Seed Speech
60
- * key lands.** The docs list the member fields flat (`speaker | audio_data |
61
- * audio_url | image_data | image_url`) without a worked example, so whether
62
- * the server wants a flat `{ speaker }` or a discriminated `{ type, ... }`
63
- * could not be settled. The adapter sends the flat reading — see
64
- * `buildTTSRequestBody` in `../adapters/tts`.
59
+ * The member fields are flat (`speaker | audio_data | audio_url | image_data |
60
+ * image_url`), confirmed against
61
+ * docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01.
65
62
  */
66
63
  export interface BytePlusTTSReference {
67
- /** Stock voice id, e.g. `en_female_stokie_uranus_bigtts`. */
64
+ /**
65
+ * Voice id, e.g. `en_female_stokie_uranus_bigtts`. Either a stock TTS 2.0
66
+ * voice or the id of a voice you cloned through Voice Replication — this
67
+ * field is the join between replication and synthesis.
68
+ */
68
69
  speaker?: string
69
70
  /** URL of a reference clip to clone (≤30 s, ≤10 MB). */
70
71
  audio_url?: string
@@ -100,6 +101,37 @@ export interface BytePlusTTSAudioConfig {
100
101
  enable_subtitle?: boolean
101
102
  }
102
103
 
104
+ /**
105
+ * `watermark` block of a TTS request.
106
+ *
107
+ * Two independent markers, both off by default:
108
+ *
109
+ * - `aigc_watermark` — **explicit**: appends an audible rhythm marker to the
110
+ * end of the clip.
111
+ * - `aigc_metadata` — **implicit**: writes provenance metadata into the audio
112
+ * header. Nothing is written unless `enable` is `true`.
113
+ *
114
+ * This is an object, not a boolean — unlike Seedream images and Seedance
115
+ * video, where `watermark` genuinely is a boolean.
116
+ */
117
+ export interface BytePlusTTSWatermark {
118
+ /** Append an audible rhythm marker to the end of the audio. Default `false`. */
119
+ aigc_watermark?: boolean
120
+ /** Provenance metadata written into the audio header. */
121
+ aigc_metadata?: {
122
+ /** Write the metadata. Default `false` — the rest is ignored without it. */
123
+ enable?: boolean
124
+ /** Name or code of the synthesis provider. */
125
+ content_producer?: string
126
+ /** Content production id. */
127
+ produce_id?: string
128
+ /** Name or code of the distribution provider. */
129
+ content_propagator?: string
130
+ /** Content distribution id. */
131
+ propagate_id?: string
132
+ }
133
+ }
134
+
103
135
  /** Request body for `POST /api/v3/tts/create` — exactly five fields. */
104
136
  export interface BytePlusTTSCreateRequest {
105
137
  /** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */
@@ -121,11 +153,8 @@ export interface BytePlusTTSCreateRequest {
121
153
  /** Voice selection and cloning references. See {@link BytePlusTTSReference}. */
122
154
  references?: Array<BytePlusTTSReference>
123
155
  audio_config?: BytePlusTTSAudioConfig
124
- /**
125
- * Watermark the generated audio. The field name is confirmed; the boolean
126
- * type is assumed by analogy with Seedream's `watermark` and unprobed.
127
- */
128
- watermark?: boolean
156
+ /** Watermark the generated audio. See {@link BytePlusTTSWatermark}. */
157
+ watermark?: BytePlusTTSWatermark
129
158
  }
130
159
 
131
160
  /**
package/src/index.ts CHANGED
@@ -47,6 +47,7 @@ export type {
47
47
  export {
48
48
  BYTEPLUS_DEFAULT_TTS_SPEAKER,
49
49
  BYTEPLUS_TTS_MAX_OUTPUT_SECONDS,
50
+ BYTEPLUS_TTS_MAX_REFERENCES,
50
51
  BytePlusTTSAdapter,
51
52
  byteplusSpeech,
52
53
  createBytePlusSpeech,
@@ -84,6 +85,7 @@ export type {
84
85
  BytePlusTTSSampleRate,
85
86
  BytePlusTTSSubtitle,
86
87
  BytePlusTTSSubtitleEntry,
88
+ BytePlusTTSWatermark,
87
89
  BytePlusVoiceErrorBody,
88
90
  } from './audio/wire-types'
89
91
  export {