@tanstack/ai-byteplus 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +202 -0
- package/dist/esm/adapters/image.d.ts +89 -0
- package/dist/esm/adapters/image.js +229 -0
- package/dist/esm/adapters/image.js.map +1 -0
- package/dist/esm/adapters/text.d.ts +163 -0
- package/dist/esm/adapters/text.js +347 -0
- package/dist/esm/adapters/text.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +102 -0
- package/dist/esm/adapters/transcription.js +274 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +143 -0
- package/dist/esm/adapters/tts.js +307 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/adapters/video.d.ts +182 -0
- package/dist/esm/adapters/video.js +442 -0
- package/dist/esm/adapters/video.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
- package/dist/esm/audio/tts-provider-options.d.ts +114 -0
- package/dist/esm/audio/wire-types.d.ts +261 -0
- package/dist/esm/audio/wire-types.js +28 -0
- package/dist/esm/audio/wire-types.js.map +1 -0
- package/dist/esm/image/image-provider-options.d.ts +165 -0
- package/dist/esm/image/image-provider-options.js +134 -0
- package/dist/esm/image/image-provider-options.js.map +1 -0
- package/dist/esm/image/wire-types.d.ts +149 -0
- package/dist/esm/index.d.ts +25 -0
- package/dist/esm/index.js +11 -0
- package/dist/esm/message-types.d.ts +154 -0
- package/dist/esm/model-meta.d.ts +594 -0
- package/dist/esm/model-meta.js +619 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/text/text-provider-options.d.ts +109 -0
- package/dist/esm/utils/client.d.ts +183 -0
- package/dist/esm/utils/client.js +253 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +197 -0
- package/dist/esm/video/video-provider-options.js +191 -0
- package/dist/esm/video/video-provider-options.js.map +1 -0
- package/dist/esm/video/wire-types.d.ts +248 -0
- package/package.json +77 -0
- package/src/adapters/image.ts +409 -0
- package/src/adapters/text.ts +539 -0
- package/src/adapters/transcription.ts +479 -0
- package/src/adapters/tts.ts +447 -0
- package/src/adapters/video.ts +732 -0
- package/src/audio/transcription-provider-options.ts +46 -0
- package/src/audio/tts-provider-options.ts +122 -0
- package/src/audio/wire-types.ts +290 -0
- package/src/image/image-provider-options.ts +288 -0
- package/src/image/wire-types.ts +169 -0
- package/src/index.ts +222 -0
- package/src/message-types.ts +169 -0
- package/src/model-meta.ts +954 -0
- package/src/text/text-provider-options.ts +151 -0
- package/src/utils/client.ts +377 -0
- package/src/video/video-provider-options.ts +361 -0
- package/src/video/wire-types.ts +293 -0
|
@@ -0,0 +1,307 @@
|
|
|
1
|
+
import { bytePlusVoiceError, bytePlusVoiceHeaders, getBytePlusVoiceApiKeyFromEnv, readJsonBody, withBytePlusVoiceDefaults } from "../utils/client.js";
|
|
2
|
+
import { BaseTTSAdapter } from "@tanstack/ai/adapters";
|
|
3
|
+
import { toRunErrorPayload } from "@tanstack/ai/adapter-internals";
|
|
4
|
+
import { generateId } from "@tanstack/ai-utils";
|
|
5
|
+
//#region src/adapters/tts.ts
|
|
6
|
+
/** Path of the synchronous Seed Speech synthesis endpoint. */
|
|
7
|
+
var TTS_CREATE_PATH = "/api/v3/tts/create";
|
|
8
|
+
/**
|
|
9
|
+
* Name of the request field carrying the text to speak.
|
|
10
|
+
*
|
|
11
|
+
* **`text_prompt` is correct — do not "fix" this to `text`.** The endpoint
|
|
12
|
+
* schema (docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01) lists this
|
|
13
|
+
* body as exactly `model`, `text_prompt`, `references`, `audio_config`,
|
|
14
|
+
* `watermark`; there is no request-side `text`. The `text` spelling belongs
|
|
15
|
+
* to the *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits
|
|
16
|
+
* under `req_params.text`. The name stays isolated here so the two spellings
|
|
17
|
+
* never get conflated.
|
|
18
|
+
*/
|
|
19
|
+
var TTS_TEXT_FIELD = "text_prompt";
|
|
20
|
+
/**
|
|
21
|
+
* Sample rate used when the caller doesn't pick one. The endpoint documents a
|
|
22
|
+
* default of 40000, which is not among the rates it accepts — so the adapter
|
|
23
|
+
* never relies on the server default and always sends this instead.
|
|
24
|
+
*/
|
|
25
|
+
var DEFAULT_SAMPLE_RATE = 24e3;
|
|
26
|
+
/**
|
|
27
|
+
* True when a Seed Speech envelope's `code` means success.
|
|
28
|
+
*
|
|
29
|
+
* Seed Speech uses `0` for success and a flat numeric code otherwise
|
|
30
|
+
* (`45000010 Invalid X-Api-Key`). The success envelope has not been confirmed
|
|
31
|
+
* against a live key, so `code` is accepted as a number, its string form, or
|
|
32
|
+
* absent — an envelope that omits `code` entirely is treated as success, which
|
|
33
|
+
* is what the HTTP status already told us.
|
|
34
|
+
*/
|
|
35
|
+
function isZeroCode(code) {
|
|
36
|
+
if (code === void 0) return true;
|
|
37
|
+
return Number(code) === 0;
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Voice used when neither `TTSOptions.voice` nor `modelOptions.speaker` is
|
|
41
|
+
* set — the English female "Stokie" voice from the TTS 2.0 generation.
|
|
42
|
+
*/
|
|
43
|
+
var BYTEPLUS_DEFAULT_TTS_SPEAKER = "en_female_stokie_uranus_bigtts";
|
|
44
|
+
/**
|
|
45
|
+
* Hard cap on a single Seed Speech synthesis, in seconds. Longer scripts must
|
|
46
|
+
* be split across calls and stitched client-side.
|
|
47
|
+
*
|
|
48
|
+
* The cap applies to the *pre-rate* length the service bills on — the
|
|
49
|
+
* `original_duration` it returns. The delivered clip can run longer than this
|
|
50
|
+
* when `speech_rate` slows it down.
|
|
51
|
+
*/
|
|
52
|
+
var BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
|
|
53
|
+
/**
|
|
54
|
+
* BytePlus Seed Speech text-to-speech adapter.
|
|
55
|
+
*
|
|
56
|
+
* Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.
|
|
57
|
+
* Two things differ from the Ark-hosted adapters in this package:
|
|
58
|
+
*
|
|
59
|
+
* - **Separate API key.** Seed Speech authenticates with `X-Api-Key` and its
|
|
60
|
+
* own key (`BYTEPLUS_VOICE_API_KEY`). An `ARK_API_KEY` sent here is
|
|
61
|
+
* rejected with `45000010 Invalid X-Api-Key`.
|
|
62
|
+
* - **120 s output cap.** A single call synthesises at most
|
|
63
|
+
* {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} seconds of audio; longer scripts
|
|
64
|
+
* have to be split and stitched client-side. The cap is measured before
|
|
65
|
+
* `speech_rate` is applied, so a slowed clip can play for longer than that
|
|
66
|
+
* — `result.duration` is the delivered length and `result.originalDuration`
|
|
67
|
+
* is the metered one.
|
|
68
|
+
*
|
|
69
|
+
* Streaming synthesis (`tts/unidirectional` and the WebSocket API) is not
|
|
70
|
+
* covered by this adapter — note that endpoint spells the text field `text`,
|
|
71
|
+
* while this one uses `text_prompt`.
|
|
72
|
+
*
|
|
73
|
+
* @example
|
|
74
|
+
* ```ts
|
|
75
|
+
* const adapter = byteplusSpeech('seed-audio-1.0')
|
|
76
|
+
* const result = await generateSpeech({
|
|
77
|
+
* adapter,
|
|
78
|
+
* text: 'welcome to the guitar store',
|
|
79
|
+
* voice: 'en_female_stokie_uranus_bigtts',
|
|
80
|
+
* format: 'mp3',
|
|
81
|
+
* })
|
|
82
|
+
* ```
|
|
83
|
+
*/
|
|
84
|
+
var BytePlusTTSAdapter = class extends BaseTTSAdapter {
|
|
85
|
+
name = "byteplus";
|
|
86
|
+
apiKey;
|
|
87
|
+
baseURL;
|
|
88
|
+
defaultHeaders;
|
|
89
|
+
fetchImpl;
|
|
90
|
+
constructor(model, config) {
|
|
91
|
+
super(model, config);
|
|
92
|
+
const resolved = withBytePlusVoiceDefaults(config);
|
|
93
|
+
this.apiKey = resolved.apiKey;
|
|
94
|
+
this.baseURL = resolved.baseURL ?? "https://voice.ap-southeast-1.bytepluses.com";
|
|
95
|
+
this.defaultHeaders = resolved.defaultHeaders ?? {};
|
|
96
|
+
this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis);
|
|
97
|
+
}
|
|
98
|
+
async generateSpeech(options) {
|
|
99
|
+
const { logger, model, text, voice, format, speed, modelOptions } = options;
|
|
100
|
+
logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {
|
|
101
|
+
provider: "byteplus",
|
|
102
|
+
model
|
|
103
|
+
});
|
|
104
|
+
const { body, audioFormat, sampleRate } = buildTTSRequestBody({
|
|
105
|
+
model,
|
|
106
|
+
text,
|
|
107
|
+
voice,
|
|
108
|
+
format,
|
|
109
|
+
speed,
|
|
110
|
+
modelOptions,
|
|
111
|
+
logger
|
|
112
|
+
});
|
|
113
|
+
try {
|
|
114
|
+
const response = await this.fetchImpl(`${this.baseURL}${TTS_CREATE_PATH}`, {
|
|
115
|
+
method: "POST",
|
|
116
|
+
headers: bytePlusVoiceHeaders(this.apiKey, {
|
|
117
|
+
...this.defaultHeaders,
|
|
118
|
+
"X-Api-Request-Id": newRequestId()
|
|
119
|
+
}),
|
|
120
|
+
body: JSON.stringify(body)
|
|
121
|
+
});
|
|
122
|
+
const payload = await readJsonBody(response);
|
|
123
|
+
if (!response.ok) throw bytePlusVoiceError(response.status, payload, "text-to-speech");
|
|
124
|
+
const data = payload;
|
|
125
|
+
if (!isZeroCode(data.code)) throw bytePlusVoiceError(response.status, payload, "text-to-speech");
|
|
126
|
+
if (typeof data.audio !== "string" || data.audio.length === 0) throw new Error(`BytePlus Seed Speech text-to-speech returned a success response with no audio (model ${model}).`);
|
|
127
|
+
const duration = toDurationSeconds(data.duration);
|
|
128
|
+
const originalDuration = toDurationSeconds(data.original_duration);
|
|
129
|
+
return {
|
|
130
|
+
id: generateId(this.name),
|
|
131
|
+
model,
|
|
132
|
+
audio: data.audio,
|
|
133
|
+
format: audioFormat,
|
|
134
|
+
contentType: getContentType(audioFormat, sampleRate),
|
|
135
|
+
...duration !== void 0 && { duration },
|
|
136
|
+
...originalDuration !== void 0 && { originalDuration },
|
|
137
|
+
...data.subtitle !== void 0 && { subtitle: data.subtitle },
|
|
138
|
+
...data.url !== void 0 && { url: data.url }
|
|
139
|
+
};
|
|
140
|
+
} catch (error) {
|
|
141
|
+
logger.errors("byteplus.generateSpeech fatal", {
|
|
142
|
+
error: toRunErrorPayload(error, "byteplus.generateSpeech failed"),
|
|
143
|
+
source: "byteplus.generateSpeech"
|
|
144
|
+
});
|
|
145
|
+
throw error;
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
};
|
|
149
|
+
/**
|
|
150
|
+
* Build the JSON body for `POST /api/v3/tts/create`, resolving the voice,
|
|
151
|
+
* output format and rate fields in one place.
|
|
152
|
+
*
|
|
153
|
+
* Returns the request `body`, the resolved `audioFormat` and the
|
|
154
|
+
* `sampleRate`, which the caller reports on the result and turns into a
|
|
155
|
+
* `contentType`.
|
|
156
|
+
*/
|
|
157
|
+
function buildTTSRequestBody(options) {
|
|
158
|
+
const { model, text, voice, format, speed, modelOptions, logger } = options;
|
|
159
|
+
const audioFormat = pickAudioFormat(modelOptions?.format, format, logger);
|
|
160
|
+
const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE;
|
|
161
|
+
const audioConfig = {
|
|
162
|
+
format: audioFormat,
|
|
163
|
+
sample_rate: sampleRate
|
|
164
|
+
};
|
|
165
|
+
if (modelOptions?.pitch_rate !== void 0) audioConfig.pitch_rate = modelOptions.pitch_rate;
|
|
166
|
+
if (modelOptions?.loudness_rate !== void 0) audioConfig.loudness_rate = modelOptions.loudness_rate;
|
|
167
|
+
if (modelOptions?.enable_subtitle !== void 0) audioConfig.enable_subtitle = modelOptions.enable_subtitle;
|
|
168
|
+
const speechRate = modelOptions?.speech_rate ?? (speed !== void 0 ? toSpeechRate(speed, logger) : void 0);
|
|
169
|
+
if (speechRate !== void 0) audioConfig.speech_rate = speechRate;
|
|
170
|
+
const body = {
|
|
171
|
+
model,
|
|
172
|
+
[TTS_TEXT_FIELD]: text,
|
|
173
|
+
references: modelOptions?.references ?? [{ speaker: modelOptions?.speaker ?? voice ?? "en_female_stokie_uranus_bigtts" }],
|
|
174
|
+
audio_config: audioConfig
|
|
175
|
+
};
|
|
176
|
+
if (modelOptions?.watermark !== void 0) body.watermark = modelOptions.watermark;
|
|
177
|
+
return {
|
|
178
|
+
body,
|
|
179
|
+
audioFormat,
|
|
180
|
+
sampleRate
|
|
181
|
+
};
|
|
182
|
+
}
|
|
183
|
+
/**
|
|
184
|
+
* Convert the cross-provider `TTSOptions.speed` multiplier into Seed Speech's
|
|
185
|
+
* `speech_rate` percentage.
|
|
186
|
+
*
|
|
187
|
+
* ```
|
|
188
|
+
* speech_rate = clamp(round((speed - 1) * 100), -50, 100)
|
|
189
|
+
* ```
|
|
190
|
+
*
|
|
191
|
+
* so `0.5 → -50`, `1.0 → 0`, `1.5 → 50`, `2.0 → 100`.
|
|
192
|
+
*
|
|
193
|
+
* The range `-50`..`100` and both multiplier anchors (`-50` = 0.5×,
|
|
194
|
+
* `100` = 2×) are documented, and hold on both the `/tts/create` and
|
|
195
|
+
* `/tts/unidirectional` endpoints.
|
|
196
|
+
*
|
|
197
|
+
* `TTSOptions.speed` spans a wider 0.25×–4× than the endpoint supports, so
|
|
198
|
+
* anything outside 0.5×–2× clamps (and warns) rather than erroring.
|
|
199
|
+
*/
|
|
200
|
+
function toSpeechRate(speed, logger) {
|
|
201
|
+
const rate = Math.round((speed - 1) * 100);
|
|
202
|
+
const clamped = Math.min(100, Math.max(-50, rate));
|
|
203
|
+
if (clamped !== rate) logger?.warn(`Speed ${speed}× is outside the range BytePlus Seed Speech documents (0.5×–2×) — clamping speech_rate from ${rate} to ${clamped}.`, {
|
|
204
|
+
provider: "byteplus",
|
|
205
|
+
requestedSpeed: speed,
|
|
206
|
+
speechRate: clamped
|
|
207
|
+
});
|
|
208
|
+
return clamped;
|
|
209
|
+
}
|
|
210
|
+
/**
|
|
211
|
+
* Map the cross-provider `TTSOptions.format` onto a Seed Speech output
|
|
212
|
+
* format. An explicit `modelOptions.format` always wins.
|
|
213
|
+
*
|
|
214
|
+
* Seed Speech produces `wav`, `mp3`, `pcm` and `ogg_opus`. The generic
|
|
215
|
+
* `opus` maps onto `ogg_opus`; `aac` and `flac` have no equivalent and fall
|
|
216
|
+
* back to `mp3` (matching how the other non-OpenAI TTS adapters in this repo
|
|
217
|
+
* handle unsupported codecs). The fallback is logged so it isn't silent.
|
|
218
|
+
*/
|
|
219
|
+
function pickAudioFormat(override, format, logger) {
|
|
220
|
+
if (override) return override;
|
|
221
|
+
if (!format) return "mp3";
|
|
222
|
+
switch (format) {
|
|
223
|
+
case "mp3":
|
|
224
|
+
case "wav":
|
|
225
|
+
case "pcm": return format;
|
|
226
|
+
case "opus": return "ogg_opus";
|
|
227
|
+
case "aac":
|
|
228
|
+
case "flac":
|
|
229
|
+
logger.warn(`BytePlus Seed Speech does not support ${format} output — falling back to mp3. Set modelOptions.format to choose between wav, mp3, pcm and ogg_opus.`, {
|
|
230
|
+
provider: "byteplus",
|
|
231
|
+
requestedFormat: format
|
|
232
|
+
});
|
|
233
|
+
return "mp3";
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
/**
|
|
237
|
+
* Build a per-request id for the `X-Api-Request-Id` header. Falls back to the
|
|
238
|
+
* package's own id generator on runtimes without `crypto.randomUUID`.
|
|
239
|
+
*/
|
|
240
|
+
function newRequestId() {
|
|
241
|
+
return globalThis.crypto?.randomUUID?.() ?? generateId("byteplus-tts");
|
|
242
|
+
}
|
|
243
|
+
/**
|
|
244
|
+
* MIME type for a Seed Speech output format.
|
|
245
|
+
*
|
|
246
|
+
* `pcm` is raw little-endian 16-bit samples, so its media type has to carry
|
|
247
|
+
* the sample rate (RFC 3551/3555). The adapter always resolves a rate, so one
|
|
248
|
+
* is always available here.
|
|
249
|
+
*/
|
|
250
|
+
function getContentType(format, sampleRate) {
|
|
251
|
+
switch (format) {
|
|
252
|
+
case "mp3": return "audio/mpeg";
|
|
253
|
+
case "wav": return "audio/wav";
|
|
254
|
+
case "ogg_opus": return "audio/ogg;codecs=opus";
|
|
255
|
+
case "pcm": return `audio/L16;rate=${sampleRate ?? 24e3}`;
|
|
256
|
+
}
|
|
257
|
+
}
|
|
258
|
+
/**
|
|
259
|
+
* Coerce a `duration` / `original_duration` field to a usable number of
|
|
260
|
+
* seconds.
|
|
261
|
+
*
|
|
262
|
+
* Both are documented as float **seconds**, so this only parses the string
|
|
263
|
+
* form and drops values that can't be a length (zero, negative, non-numeric).
|
|
264
|
+
* Note that `duration` may legitimately exceed
|
|
265
|
+
* {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} — a clip synthesised at
|
|
266
|
+
* `speech_rate: -50` runs at half speed, so up to 120 s of billed audio can
|
|
267
|
+
* be delivered as up to 240 s of playback. Do not "correct" a large value
|
|
268
|
+
* here; the cap applies to `original_duration`.
|
|
269
|
+
*
|
|
270
|
+
* The subtitle timings are the mixed-unit exception: those are milliseconds
|
|
271
|
+
* and are passed through untouched on {@link BytePlusTTSResult.subtitle}.
|
|
272
|
+
*/
|
|
273
|
+
function toDurationSeconds(raw) {
|
|
274
|
+
const value = typeof raw === "string" ? Number(raw) : raw;
|
|
275
|
+
if (value === void 0 || !Number.isFinite(value) || value <= 0) return;
|
|
276
|
+
return value;
|
|
277
|
+
}
|
|
278
|
+
/**
|
|
279
|
+
* Creates a BytePlus Seed Speech TTS adapter with an explicit API key.
|
|
280
|
+
*
|
|
281
|
+
* The key is the **Seed Speech** key, not the Ark key used by the chat, image
|
|
282
|
+
* and video adapters.
|
|
283
|
+
*
|
|
284
|
+
* @example
|
|
285
|
+
* ```ts
|
|
286
|
+
* const adapter = createBytePlusSpeech('seed-audio-1.0', process.env.BYTEPLUS_VOICE_API_KEY!)
|
|
287
|
+
* ```
|
|
288
|
+
*/
|
|
289
|
+
function createBytePlusSpeech(model, apiKey, config) {
|
|
290
|
+
return new BytePlusTTSAdapter(model, {
|
|
291
|
+
...config,
|
|
292
|
+
apiKey
|
|
293
|
+
});
|
|
294
|
+
}
|
|
295
|
+
/**
|
|
296
|
+
* Creates a BytePlus Seed Speech TTS adapter, reading the API key from
|
|
297
|
+
* `BYTEPLUS_VOICE_API_KEY`.
|
|
298
|
+
*
|
|
299
|
+
* @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.
|
|
300
|
+
*/
|
|
301
|
+
function byteplusSpeech(model, config) {
|
|
302
|
+
return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config);
|
|
303
|
+
}
|
|
304
|
+
//#endregion
|
|
305
|
+
export { BYTEPLUS_DEFAULT_TTS_SPEAKER, BYTEPLUS_TTS_MAX_OUTPUT_SECONDS, BytePlusTTSAdapter, buildTTSRequestBody, byteplusSpeech, createBytePlusSpeech, getContentType, toDurationSeconds, toSpeechRate };
|
|
306
|
+
|
|
307
|
+
//# sourceMappingURL=tts.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tts.js","names":[],"sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport { generateId } from '@tanstack/ai-utils'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport {\n BYTEPLUS_VOICE_BASE_URL,\n bytePlusVoiceError,\n bytePlusVoiceHeaders,\n getBytePlusVoiceApiKeyFromEnv,\n readJsonBody,\n withBytePlusVoiceDefaults,\n} from '../utils/client'\nimport type { TTSOptions } from '@tanstack/ai'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { BytePlusVoiceConfig } from '../utils/client'\nimport type { BytePlusTTSModel } from '../model-meta'\nimport type {\n BytePlusTTSAudioConfig,\n BytePlusTTSAudioFormat,\n BytePlusTTSCreateRequest,\n BytePlusTTSCreateResponse,\n} from '../audio/wire-types'\nimport type {\n BytePlusTTSProviderOptions,\n BytePlusTTSResult,\n} from '../audio/tts-provider-options'\n\n/** Path of the synchronous Seed Speech synthesis endpoint. */\nconst TTS_CREATE_PATH = '/api/v3/tts/create'\n\n/**\n * Name of the request field carrying the text to speak.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** The endpoint\n * schema (docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01) lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`; there is no request-side `text`. The `text` spelling belongs\n * to the *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits\n * under `req_params.text`. The name stays isolated here so the two spellings\n * never get conflated.\n */\nconst TTS_TEXT_FIELD = 'text_prompt' satisfies keyof BytePlusTTSCreateRequest\n\n/**\n * Sample rate used when the caller doesn't pick one. The endpoint documents a\n * default of 40000, which is not among the rates it accepts — so the adapter\n * never relies on the server default and always sends this instead.\n */\nconst DEFAULT_SAMPLE_RATE = 24000\n\n/**\n * True when a Seed Speech envelope's `code` means success.\n *\n * Seed Speech uses `0` for success and a flat numeric code otherwise\n * (`45000010 Invalid X-Api-Key`). The success envelope has not been confirmed\n * against a live key, so `code` is accepted as a number, its string form, or\n * absent — an envelope that omits `code` entirely is treated as success, which\n * is what the HTTP status already told us.\n */\nfunction isZeroCode(code: number | string | undefined): boolean {\n if (code === undefined) return true\n return Number(code) === 0\n}\n\n/**\n * Voice used when neither `TTSOptions.voice` nor `modelOptions.speaker` is\n * set — the English female \"Stokie\" voice from the TTS 2.0 generation.\n */\nexport const BYTEPLUS_DEFAULT_TTS_SPEAKER = 'en_female_stokie_uranus_bigtts'\n\n/**\n * Hard cap on a single Seed Speech synthesis, in seconds. Longer scripts must\n * be split across calls and stitched client-side.\n *\n * The cap applies to the *pre-rate* length the service bills on — the\n * `original_duration` it returns. The delivered clip can run longer than this\n * when `speech_rate` slows it down.\n */\nexport const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120\n\n/**\n * BytePlus Seed Speech text-to-speech adapter.\n *\n * Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.\n * Two things differ from the Ark-hosted adapters in this package:\n *\n * - **Separate API key.** Seed Speech authenticates with `X-Api-Key` and its\n * own key (`BYTEPLUS_VOICE_API_KEY`). An `ARK_API_KEY` sent here is\n * rejected with `45000010 Invalid X-Api-Key`.\n * - **120 s output cap.** A single call synthesises at most\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} seconds of audio; longer scripts\n * have to be split and stitched client-side. The cap is measured before\n * `speech_rate` is applied, so a slowed clip can play for longer than that\n * — `result.duration` is the delivered length and `result.originalDuration`\n * is the metered one.\n *\n * Streaming synthesis (`tts/unidirectional` and the WebSocket API) is not\n * covered by this adapter — note that endpoint spells the text field `text`,\n * while this one uses `text_prompt`.\n *\n * @example\n * ```ts\n * const adapter = byteplusSpeech('seed-audio-1.0')\n * const result = await generateSpeech({\n * adapter,\n * text: 'welcome to the guitar store',\n * voice: 'en_female_stokie_uranus_bigtts',\n * format: 'mp3',\n * })\n * ```\n */\nexport class BytePlusTTSAdapter<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {\n readonly name = 'byteplus' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n private readonly fetchImpl: typeof fetch\n\n constructor(model: TModel, config: BytePlusVoiceConfig) {\n super(model, config)\n const resolved = withBytePlusVoiceDefaults(config)\n this.apiKey = resolved.apiKey\n this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL\n this.defaultHeaders = resolved.defaultHeaders ?? {}\n this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)\n }\n\n async generateSpeech(\n options: TTSOptions<BytePlusTTSProviderOptions>,\n ): Promise<BytePlusTTSResult> {\n const { logger, model, text, voice, format, speed, modelOptions } = options\n\n logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {\n provider: 'byteplus',\n model,\n })\n\n const { body, audioFormat, sampleRate } = buildTTSRequestBody({\n model,\n text,\n voice,\n format,\n speed,\n modelOptions,\n logger,\n })\n\n try {\n const response = await this.fetchImpl(\n `${this.baseURL}${TTS_CREATE_PATH}`,\n {\n method: 'POST',\n headers: bytePlusVoiceHeaders(this.apiKey, {\n ...this.defaultHeaders,\n // Client-generated per-request id. BytePlus echoes it in their\n // request logs, which is what support asks for when diagnosing a\n // synthesis failure.\n 'X-Api-Request-Id': newRequestId(),\n }),\n body: JSON.stringify(body),\n },\n )\n\n const payload = await readJsonBody(response)\n\n if (!response.ok) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n const data = payload as BytePlusTTSCreateResponse\n\n // Seed Speech reports status in the body, not only in the HTTP status:\n // a 200 can carry a non-zero `code`. Check it before looking at `audio`,\n // because a failed call may still return a partial or placeholder\n // payload that would otherwise be handed back as if it were valid.\n //\n // `code` is accepted as a number *or* a string. The success envelope was\n // never confirmed against a live key (no voice key yet — see\n // `audio/wire-types.ts`), and `readStringField` already tolerates both\n // forms when rendering the error, so requiring a number here would let\n // `{\"code\": \"45000010\"}` through both this gate and the one below.\n if (!isZeroCode(data.code)) {\n throw bytePlusVoiceError(response.status, payload, 'text-to-speech')\n }\n\n // Belt and braces for a 200 that reports success but carries nothing to\n // play. Say that the adapter rejected it, rather than reusing the\n // envelope-error phrasing — a bare \"failed (200)\" gives no hint that the\n // response was well-formed and simply empty.\n if (typeof data.audio !== 'string' || data.audio.length === 0) {\n throw new Error(\n `BytePlus Seed Speech text-to-speech returned a success response ` +\n `with no audio (model ${model}).`,\n )\n }\n\n const duration = toDurationSeconds(data.duration)\n const originalDuration = toDurationSeconds(data.original_duration)\n\n return {\n id: generateId(this.name),\n model,\n audio: data.audio,\n format: audioFormat,\n contentType: getContentType(audioFormat, sampleRate),\n ...(duration !== undefined && { duration }),\n ...(originalDuration !== undefined && { originalDuration }),\n ...(data.subtitle !== undefined && { subtitle: data.subtitle }),\n ...(data.url !== undefined && { url: data.url }),\n }\n } catch (error) {\n logger.errors('byteplus.generateSpeech fatal', {\n error: toRunErrorPayload(error, 'byteplus.generateSpeech failed'),\n source: 'byteplus.generateSpeech',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the JSON body for `POST /api/v3/tts/create`, resolving the voice,\n * output format and rate fields in one place.\n *\n * Returns the request `body`, the resolved `audioFormat` and the\n * `sampleRate`, which the caller reports on the result and turns into a\n * `contentType`.\n */\nexport function buildTTSRequestBody(options: {\n model: string\n text: string\n voice: string | undefined\n format: TTSOptions['format'] | undefined\n speed: number | undefined\n modelOptions: BytePlusTTSProviderOptions | undefined\n logger: InternalLogger\n}): {\n body: BytePlusTTSCreateRequest\n audioFormat: BytePlusTTSAudioFormat\n sampleRate: number\n} {\n const { model, text, voice, format, speed, modelOptions, logger } = options\n\n const audioFormat = pickAudioFormat(modelOptions?.format, format, logger)\n // Always explicit: the documented server default (40000) is not one of the\n // rates the endpoint accepts, so relying on it is a coin flip.\n const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE\n\n const audioConfig: BytePlusTTSAudioConfig = {\n format: audioFormat,\n sample_rate: sampleRate,\n }\n if (modelOptions?.pitch_rate !== undefined) {\n audioConfig.pitch_rate = modelOptions.pitch_rate\n }\n if (modelOptions?.loudness_rate !== undefined) {\n audioConfig.loudness_rate = modelOptions.loudness_rate\n }\n if (modelOptions?.enable_subtitle !== undefined) {\n audioConfig.enable_subtitle = modelOptions.enable_subtitle\n }\n\n // An explicit `speech_rate` always wins over the derived one — it is the\n // native unit and the only way to reach the extremes precisely.\n const speechRate =\n modelOptions?.speech_rate ??\n (speed !== undefined ? toSpeechRate(speed, logger) : undefined)\n if (speechRate !== undefined) {\n audioConfig.speech_rate = speechRate\n }\n\n const body: BytePlusTTSCreateRequest = {\n model,\n [TTS_TEXT_FIELD]: text,\n // The voice belongs inside `references`, not at the top level — a\n // top-level `speaker` is silently ignored by the server. The flat member\n // shape here is the best-supported reading of the docs; see\n // `BytePlusTTSReference` for the unresolved part and the live-probe flag.\n references: modelOptions?.references ?? [\n {\n speaker: modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,\n },\n ],\n audio_config: audioConfig,\n }\n if (modelOptions?.watermark !== undefined) {\n body.watermark = modelOptions.watermark\n }\n\n return { body, audioFormat, sampleRate }\n}\n\n/**\n * Convert the cross-provider `TTSOptions.speed` multiplier into Seed Speech's\n * `speech_rate` percentage.\n *\n * ```\n * speech_rate = clamp(round((speed - 1) * 100), -50, 100)\n * ```\n *\n * so `0.5 → -50`, `1.0 → 0`, `1.5 → 50`, `2.0 → 100`.\n *\n * The range `-50`..`100` and both multiplier anchors (`-50` = 0.5×,\n * `100` = 2×) are documented, and hold on both the `/tts/create` and\n * `/tts/unidirectional` endpoints.\n *\n * `TTSOptions.speed` spans a wider 0.25×–4× than the endpoint supports, so\n * anything outside 0.5×–2× clamps (and warns) rather than erroring.\n */\nexport function toSpeechRate(speed: number, logger?: InternalLogger): number {\n const rate = Math.round((speed - 1) * 100)\n const clamped = Math.min(100, Math.max(-50, rate))\n if (clamped !== rate) {\n logger?.warn(\n `Speed ${speed}× is outside the range BytePlus Seed Speech documents (0.5×–2×) — clamping speech_rate from ${rate} to ${clamped}.`,\n { provider: 'byteplus', requestedSpeed: speed, speechRate: clamped },\n )\n }\n return clamped\n}\n\n/**\n * Map the cross-provider `TTSOptions.format` onto a Seed Speech output\n * format. An explicit `modelOptions.format` always wins.\n *\n * Seed Speech produces `wav`, `mp3`, `pcm` and `ogg_opus`. The generic\n * `opus` maps onto `ogg_opus`; `aac` and `flac` have no equivalent and fall\n * back to `mp3` (matching how the other non-OpenAI TTS adapters in this repo\n * handle unsupported codecs). The fallback is logged so it isn't silent.\n */\nfunction pickAudioFormat(\n override: BytePlusTTSAudioFormat | undefined,\n format: TTSOptions['format'] | undefined,\n logger: InternalLogger,\n): BytePlusTTSAudioFormat {\n if (override) return override\n if (!format) return 'mp3'\n switch (format) {\n case 'mp3':\n case 'wav':\n case 'pcm':\n return format\n case 'opus':\n return 'ogg_opus'\n case 'aac':\n case 'flac':\n logger.warn(\n `BytePlus Seed Speech does not support ${format} output — falling back to mp3. Set modelOptions.format to choose between wav, mp3, pcm and ogg_opus.`,\n { provider: 'byteplus', requestedFormat: format },\n )\n return 'mp3'\n }\n}\n\n/**\n * Build a per-request id for the `X-Api-Request-Id` header. Falls back to the\n * package's own id generator on runtimes without `crypto.randomUUID`.\n */\nfunction newRequestId(): string {\n return globalThis.crypto?.randomUUID?.() ?? generateId('byteplus-tts')\n}\n\n/**\n * MIME type for a Seed Speech output format.\n *\n * `pcm` is raw little-endian 16-bit samples, so its media type has to carry\n * the sample rate (RFC 3551/3555). The adapter always resolves a rate, so one\n * is always available here.\n */\nexport function getContentType(\n format: BytePlusTTSAudioFormat,\n sampleRate?: number,\n): string {\n switch (format) {\n case 'mp3':\n return 'audio/mpeg'\n case 'wav':\n return 'audio/wav'\n case 'ogg_opus':\n return 'audio/ogg;codecs=opus'\n case 'pcm':\n return `audio/L16;rate=${sampleRate ?? 24000}`\n }\n}\n\n/**\n * Coerce a `duration` / `original_duration` field to a usable number of\n * seconds.\n *\n * Both are documented as float **seconds**, so this only parses the string\n * form and drops values that can't be a length (zero, negative, non-numeric).\n * Note that `duration` may legitimately exceed\n * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} — a clip synthesised at\n * `speech_rate: -50` runs at half speed, so up to 120 s of billed audio can\n * be delivered as up to 240 s of playback. Do not \"correct\" a large value\n * here; the cap applies to `original_duration`.\n *\n * The subtitle timings are the mixed-unit exception: those are milliseconds\n * and are passed through untouched on {@link BytePlusTTSResult.subtitle}.\n */\nexport function toDurationSeconds(\n raw: number | string | undefined,\n): number | undefined {\n const value = typeof raw === 'string' ? Number(raw) : raw\n if (value === undefined || !Number.isFinite(value) || value <= 0) {\n return undefined\n }\n return value\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter with an explicit API key.\n *\n * The key is the **Seed Speech** key, not the Ark key used by the chat, image\n * and video adapters.\n *\n * @example\n * ```ts\n * const adapter = createBytePlusSpeech('seed-audio-1.0', process.env.BYTEPLUS_VOICE_API_KEY!)\n * ```\n */\nexport function createBytePlusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n apiKey: string,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return new BytePlusTTSAdapter(model, { ...config, apiKey })\n}\n\n/**\n * Creates a BytePlus Seed Speech TTS adapter, reading the API key from\n * `BYTEPLUS_VOICE_API_KEY`.\n *\n * @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.\n */\nexport function byteplusSpeech<\n TModel extends BytePlusTTSModel = BytePlusTTSModel,\n>(\n model: TModel,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTTSAdapter<TModel> {\n return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config)\n}\n"],"mappings":";;;;;;AA2BA,IAAM,kBAAkB;;;;;;;;;;;;AAaxB,IAAM,iBAAiB;;;;;;AAOvB,IAAM,sBAAsB;;;;;;;;;;AAW5B,SAAS,WAAW,MAA4C;CAC9D,IAAI,SAAS,KAAA,GAAW,OAAO;CAC/B,OAAO,OAAO,IAAI,MAAM;AAC1B;;;;;AAMA,IAAa,+BAA+B;;;;;;;;;AAU5C,IAAa,kCAAkC;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiC/C,IAAa,qBAAb,cAEU,eAAmD;CAC3D,OAAgB;CAEhB;CACA;CACA;CACA;CAEA,YAAY,OAAe,QAA6B;EACtD,MAAM,OAAO,MAAM;EACnB,MAAM,WAAW,0BAA0B,MAAM;EACjD,KAAK,SAAS,SAAS;EACvB,KAAK,UAAU,SAAS,WAAA;EACxB,KAAK,iBAAiB,SAAS,kBAAkB,CAAC;EAClD,KAAK,YAAY,SAAS,SAAS,WAAW,MAAM,KAAK,UAAU;CACrE;CAEA,MAAM,eACJ,SAC4B;EAC5B,MAAM,EAAE,QAAQ,OAAO,MAAM,OAAO,QAAQ,OAAO,iBAAiB;EAEpE,OAAO,QAAQ,mDAAmD,SAAS;GACzE,UAAU;GACV;EACF,CAAC;EAED,MAAM,EAAE,MAAM,aAAa,eAAe,oBAAoB;GAC5D;GACA;GACA;GACA;GACA;GACA;GACA;EACF,CAAC;EAED,IAAI;GACF,MAAM,WAAW,MAAM,KAAK,UAC1B,GAAG,KAAK,UAAU,mBAClB;IACE,QAAQ;IACR,SAAS,qBAAqB,KAAK,QAAQ;KACzC,GAAG,KAAK;KAIR,oBAAoB,aAAa;IACnC,CAAC;IACD,MAAM,KAAK,UAAU,IAAI;GAC3B,CACF;GAEA,MAAM,UAAU,MAAM,aAAa,QAAQ;GAE3C,IAAI,CAAC,SAAS,IACZ,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAGrE,MAAM,OAAO;GAYb,IAAI,CAAC,WAAW,KAAK,IAAI,GACvB,MAAM,mBAAmB,SAAS,QAAQ,SAAS,gBAAgB;GAOrE,IAAI,OAAO,KAAK,UAAU,YAAY,KAAK,MAAM,WAAW,GAC1D,MAAM,IAAI,MACR,wFAC0B,MAAM,GAClC;GAGF,MAAM,WAAW,kBAAkB,KAAK,QAAQ;GAChD,MAAM,mBAAmB,kBAAkB,KAAK,iBAAiB;GAEjE,OAAO;IACL,IAAI,WAAW,KAAK,IAAI;IACxB;IACA,OAAO,KAAK;IACZ,QAAQ;IACR,aAAa,eAAe,aAAa,UAAU;IACnD,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;IACzC,GAAI,qBAAqB,KAAA,KAAa,EAAE,iBAAiB;IACzD,GAAI,KAAK,aAAa,KAAA,KAAa,EAAE,UAAU,KAAK,SAAS;IAC7D,GAAI,KAAK,QAAQ,KAAA,KAAa,EAAE,KAAK,KAAK,IAAI;GAChD;EACF,SAAS,OAAO;GACd,OAAO,OAAO,iCAAiC;IAC7C,OAAO,kBAAkB,OAAO,gCAAgC;IAChE,QAAQ;GACV,CAAC;GACD,MAAM;EACR;CACF;AACF;;;;;;;;;AAUA,SAAgB,oBAAoB,SAYlC;CACA,MAAM,EAAE,OAAO,MAAM,OAAO,QAAQ,OAAO,cAAc,WAAW;CAEpE,MAAM,cAAc,gBAAgB,cAAc,QAAQ,QAAQ,MAAM;CAGxE,MAAM,aAAa,cAAc,eAAe;CAEhD,MAAM,cAAsC;EAC1C,QAAQ;EACR,aAAa;CACf;CACA,IAAI,cAAc,eAAe,KAAA,GAC/B,YAAY,aAAa,aAAa;CAExC,IAAI,cAAc,kBAAkB,KAAA,GAClC,YAAY,gBAAgB,aAAa;CAE3C,IAAI,cAAc,oBAAoB,KAAA,GACpC,YAAY,kBAAkB,aAAa;CAK7C,MAAM,aACJ,cAAc,gBACb,UAAU,KAAA,IAAY,aAAa,OAAO,MAAM,IAAI,KAAA;CACvD,IAAI,eAAe,KAAA,GACjB,YAAY,cAAc;CAG5B,MAAM,OAAiC;EACrC;GACC,iBAAiB;EAKlB,YAAY,cAAc,cAAc,CACtC,EACE,SAAS,cAAc,WAAW,SAAA,iCACpC,CACF;EACA,cAAc;CAChB;CACA,IAAI,cAAc,cAAc,KAAA,GAC9B,KAAK,YAAY,aAAa;CAGhC,OAAO;EAAE;EAAM;EAAa;CAAW;AACzC;;;;;;;;;;;;;;;;;;AAmBA,SAAgB,aAAa,OAAe,QAAiC;CAC3E,MAAM,OAAO,KAAK,OAAO,QAAQ,KAAK,GAAG;CACzC,MAAM,UAAU,KAAK,IAAI,KAAK,KAAK,IAAI,KAAK,IAAI,CAAC;CACjD,IAAI,YAAY,MACd,QAAQ,KACN,SAAS,MAAM,8FAA8F,KAAK,MAAM,QAAQ,IAChI;EAAE,UAAU;EAAY,gBAAgB;EAAO,YAAY;CAAQ,CACrE;CAEF,OAAO;AACT;;;;;;;;;;AAWA,SAAS,gBACP,UACA,QACA,QACwB;CACxB,IAAI,UAAU,OAAO;CACrB,IAAI,CAAC,QAAQ,OAAO;CACpB,QAAQ,QAAR;EACE,KAAK;EACL,KAAK;EACL,KAAK,OACH,OAAO;EACT,KAAK,QACH,OAAO;EACT,KAAK;EACL,KAAK;GACH,OAAO,KACL,yCAAyC,OAAO,uGAChD;IAAE,UAAU;IAAY,iBAAiB;GAAO,CAClD;GACA,OAAO;CACX;AACF;;;;;AAMA,SAAS,eAAuB;CAC9B,OAAO,WAAW,QAAQ,aAAa,KAAK,WAAW,cAAc;AACvE;;;;;;;;AASA,SAAgB,eACd,QACA,YACQ;CACR,QAAQ,QAAR;EACE,KAAK,OACH,OAAO;EACT,KAAK,OACH,OAAO;EACT,KAAK,YACH,OAAO;EACT,KAAK,OACH,OAAO,kBAAkB,cAAc;CAC3C;AACF;;;;;;;;;;;;;;;;AAiBA,SAAgB,kBACd,KACoB;CACpB,MAAM,QAAQ,OAAO,QAAQ,WAAW,OAAO,GAAG,IAAI;CACtD,IAAI,UAAU,KAAA,KAAa,CAAC,OAAO,SAAS,KAAK,KAAK,SAAS,GAC7D;CAEF,OAAO;AACT;;;;;;;;;;;;AAaA,SAAgB,qBAGd,OACA,QACA,QAC4B;CAC5B,OAAO,IAAI,mBAAmB,OAAO;EAAE,GAAG;EAAQ;CAAO,CAAC;AAC5D;;;;;;;AAQA,SAAgB,eAGd,OACA,QAC4B;CAC5B,OAAO,qBAAqB,OAAO,8BAA8B,GAAG,MAAM;AAC5E"}
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
import { BaseVideoAdapter, DurationOptions } from '@tanstack/ai/adapters';
|
|
2
|
+
import { VideoGenerationOptions, VideoJobResult, VideoStatusResult, VideoUrlResult } from '@tanstack/ai';
|
|
3
|
+
import { BytePlusVideoTaskStatus } from '../video/wire-types.js';
|
|
4
|
+
import { BytePlusVideoProviderOptions } from '../video/video-provider-options.js';
|
|
5
|
+
import { BytePlusVideoModelOrString, ResolveBytePlusVideoInputModalities, ResolveBytePlusVideoSize } from '../model-meta.js';
|
|
6
|
+
import { BytePlusArkConfig } from '../utils/client.js';
|
|
7
|
+
/**
|
|
8
|
+
* Configuration for the BytePlus Seedance video adapter.
|
|
9
|
+
*
|
|
10
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
11
|
+
*/
|
|
12
|
+
export interface BytePlusVideoConfig extends BytePlusArkConfig {
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* BytePlus Seedance video generation adapter.
|
|
16
|
+
*
|
|
17
|
+
* Drives Ark's asynchronous task API — `POST /contents/generations/tasks` to
|
|
18
|
+
* submit, `GET /contents/generations/tasks/{id}` to poll and to read the
|
|
19
|
+
* finished video URL. Core owns the polling loop; this adapter implements the
|
|
20
|
+
* three primitives plus the duration metadata.
|
|
21
|
+
*
|
|
22
|
+
* Prompt parts map onto Seedance's `content[]` roles, which the API sorts into
|
|
23
|
+
* mutually exclusive task types:
|
|
24
|
+
*
|
|
25
|
+
* - `'start_frame'` (or a single un-roled image) → `first_frame` — the frame
|
|
26
|
+
* the video opens on (`i2v`).
|
|
27
|
+
* - `'end_frame'` → `last_frame` — the frame it closes on (`flf2v`); Seedance
|
|
28
|
+
* requires a `first_frame` alongside it, and
|
|
29
|
+
* `seedance-1-0-pro-fast-251015` does not support it at all.
|
|
30
|
+
* - `'reference'` / `'character'` → `reference_image`, video parts →
|
|
31
|
+
* `reference_video`, audio parts → `reference_audio` — subject and style
|
|
32
|
+
* references the model draws on (`r2v`, Seedance 2.0 family only).
|
|
33
|
+
*
|
|
34
|
+
* Frame roles and reference roles cannot be combined in one request, so the
|
|
35
|
+
* adapter rejects a mix up front rather than surfacing a raw 400.
|
|
36
|
+
*
|
|
37
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
38
|
+
*
|
|
39
|
+
* @example
|
|
40
|
+
* ```typescript
|
|
41
|
+
* const adapter = byteplusVideo('seedance-1-0-pro-fast-251015')
|
|
42
|
+
*
|
|
43
|
+
* const { jobId } = await generateVideo({
|
|
44
|
+
* adapter,
|
|
45
|
+
* prompt: 'a guitar being played in a store',
|
|
46
|
+
* size: '16:9_720p',
|
|
47
|
+
* duration: 4,
|
|
48
|
+
* modelOptions: { service_tier: 'flex' },
|
|
49
|
+
* })
|
|
50
|
+
* ```
|
|
51
|
+
*/
|
|
52
|
+
export declare class BytePlusVideoAdapter<TModel extends BytePlusVideoModelOrString> extends BaseVideoAdapter<TModel, BytePlusVideoProviderOptions, Record<TModel, BytePlusVideoProviderOptions>, Record<TModel, ResolveBytePlusVideoSize<TModel>>, Record<TModel, ResolveBytePlusVideoInputModalities<TModel>>, Record<TModel, number>> {
|
|
53
|
+
readonly name: "byteplus";
|
|
54
|
+
/** Config with the Ark base URL resolved and its trailing slashes trimmed. */
|
|
55
|
+
private readonly clientConfig;
|
|
56
|
+
constructor(config: BytePlusVideoConfig, model: TModel);
|
|
57
|
+
private request;
|
|
58
|
+
/**
|
|
59
|
+
* Builds the `content[]` array from the resolved prompt, enforcing
|
|
60
|
+
* Seedance's role vocabulary and mode exclusivity.
|
|
61
|
+
*/
|
|
62
|
+
private buildContent;
|
|
63
|
+
createVideoJob(options: VideoGenerationOptions<BytePlusVideoProviderOptions, ResolveBytePlusVideoSize<TModel>, number>): Promise<VideoJobResult>;
|
|
64
|
+
/**
|
|
65
|
+
* Fetches a task, tagging the thrown error with the HTTP status.
|
|
66
|
+
*
|
|
67
|
+
* The 200 body is validated rather than cast. `readJsonBody` returns
|
|
68
|
+
* `undefined` for an empty body and the raw text for a non-JSON one — both
|
|
69
|
+
* documented failure modes of these hosts (an HTML error page from a proxy
|
|
70
|
+
* in front of the API). Casting either to `BytePlusVideoTask` yields a task
|
|
71
|
+
* whose `status` is `undefined`, which {@link mapStatus} would have to
|
|
72
|
+
* interpret; the honest answer is that the response was not a task at all,
|
|
73
|
+
* so say so while the body is still in hand.
|
|
74
|
+
*/
|
|
75
|
+
private retrieveTask;
|
|
76
|
+
getVideoStatus(jobId: string): Promise<VideoStatusResult>;
|
|
77
|
+
getVideoUrl(jobId: string): Promise<VideoUrlResult>;
|
|
78
|
+
/**
|
|
79
|
+
* Maps Seedance task states onto the generic video status set. `expired`
|
|
80
|
+
* (the task outlived `execution_expires_after`) and `cancelled` are
|
|
81
|
+
* terminal non-successes, so both report as failed.
|
|
82
|
+
*
|
|
83
|
+
* An unrecognized state throws rather than defaulting to `processing`.
|
|
84
|
+
* Core's poll loop treats `processing` as "keep waiting", so mapping an
|
|
85
|
+
* unknown state — a missing `status`, or a terminal one Ark adds later such
|
|
86
|
+
* as `rejected` — onto it means polling until `maxDuration` and then
|
|
87
|
+
* reporting a generic timeout, with the state Ark actually sent never
|
|
88
|
+
* reaching the caller. Failing here names it.
|
|
89
|
+
*/
|
|
90
|
+
protected mapStatus(apiStatus: BytePlusVideoTaskStatus | string | undefined): VideoStatusResult['status'];
|
|
91
|
+
/**
|
|
92
|
+
* Seedance accepts any whole second inside a per-model range: 4–15s on the
|
|
93
|
+
* 2.0 family, 4–12s on 1.5-pro, 2–12s on the 1.0 models. An unknown model
|
|
94
|
+
* reports the union of those ranges as a UI hint — see
|
|
95
|
+
* `BYTEPLUS_VIDEO_FALLBACK_DURATIONS`, which `createVideoJob` does not snap
|
|
96
|
+
* against.
|
|
97
|
+
*/
|
|
98
|
+
availableDurations(): DurationOptions<number>;
|
|
99
|
+
/**
|
|
100
|
+
* Coerce a raw seconds value to the closest duration this model accepts
|
|
101
|
+
* (clamped to its range and rounded to whole seconds).
|
|
102
|
+
*/
|
|
103
|
+
snapDuration(seconds: number): number | undefined;
|
|
104
|
+
}
|
|
105
|
+
/**
|
|
106
|
+
* Creates a BytePlus Seedance video adapter with an explicit API key.
|
|
107
|
+
* Type resolution happens here at the call site.
|
|
108
|
+
*
|
|
109
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
110
|
+
*
|
|
111
|
+
* @param model - The model name (e.g., 'seedance-1-0-pro-fast-251015')
|
|
112
|
+
* @param apiKey - Your BytePlus Ark API key
|
|
113
|
+
* @param config - Optional additional configuration
|
|
114
|
+
* @returns Configured BytePlus video adapter instance with resolved types
|
|
115
|
+
*
|
|
116
|
+
* @example
|
|
117
|
+
* ```typescript
|
|
118
|
+
* const adapter = createBytePlusVideo('seedance-1-5-pro-251215', 'ark-...')
|
|
119
|
+
*
|
|
120
|
+
* const { jobId } = await generateVideo({
|
|
121
|
+
* adapter,
|
|
122
|
+
* prompt: 'a guitar being played in a store',
|
|
123
|
+
* size: '16:9_1080p',
|
|
124
|
+
* duration: 5,
|
|
125
|
+
* })
|
|
126
|
+
* ```
|
|
127
|
+
*/
|
|
128
|
+
export declare function createBytePlusVideo<TModel extends BytePlusVideoModelOrString>(model: TModel, apiKey: string, config?: Omit<BytePlusVideoConfig, 'apiKey'>): BytePlusVideoAdapter<TModel>;
|
|
129
|
+
/**
|
|
130
|
+
* Creates a BytePlus Seedance video adapter, reading `ARK_API_KEY` from the
|
|
131
|
+
* environment. Type resolution happens here at the call site.
|
|
132
|
+
*
|
|
133
|
+
* Note that Ark keys are region-isolated and Seedance is only served from the
|
|
134
|
+
* Asia-Pacific endpoint — an EU key will not work here.
|
|
135
|
+
*
|
|
136
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
137
|
+
*
|
|
138
|
+
* @param model - The model name (e.g., 'dreamina-seedance-2-0-260128')
|
|
139
|
+
* @param config - Optional configuration (excluding apiKey, auto-detected)
|
|
140
|
+
* @returns Configured BytePlus video adapter instance with resolved types
|
|
141
|
+
* @throws Error if ARK_API_KEY is not found in environment
|
|
142
|
+
*
|
|
143
|
+
* @example
|
|
144
|
+
* ```typescript
|
|
145
|
+
* const adapter = byteplusVideo('dreamina-seedance-2-0-260128')
|
|
146
|
+
*
|
|
147
|
+
* // Image-to-video: an un-roled image is the opening frame.
|
|
148
|
+
* const { jobId } = await generateVideo({
|
|
149
|
+
* adapter,
|
|
150
|
+
* prompt: [
|
|
151
|
+
* { type: 'text', content: 'the guitarist starts playing' },
|
|
152
|
+
* { type: 'image', source: { type: 'url', value: 'https://example.com/shop.jpg' } },
|
|
153
|
+
* ],
|
|
154
|
+
* })
|
|
155
|
+
*
|
|
156
|
+
* const status = await getVideoJobStatus({ adapter, jobId })
|
|
157
|
+
* ```
|
|
158
|
+
*
|
|
159
|
+
* ## Models this package does not know yet
|
|
160
|
+
*
|
|
161
|
+
* `model` also accepts any string, so a Seedance id BytePlus publishes after
|
|
162
|
+
* this release works without upgrading. **Seedance 2.5 is the case this exists
|
|
163
|
+
* for**: `dreamina-seedance-2-5-260628` is real and reachable, but its
|
|
164
|
+
* capability cells could not be probed from this repo's account (Ark answers
|
|
165
|
+
* 404 `ModelNotOpen` until the model is activated in the Ark Console), so it
|
|
166
|
+
* is deliberately absent from the narrowed model tables. Passing it here works
|
|
167
|
+
* for an account that has activated it.
|
|
168
|
+
*
|
|
169
|
+
* An unknown id relaxes both halves of the adapter: the `size` type widens to
|
|
170
|
+
* any string, provider options are ungated, and the runtime guards that encode
|
|
171
|
+
* per-model capabilities — resolution tiers, closing-frame and reference-media
|
|
172
|
+
* support, frame cardinality and mode exclusivity, duration snapping — stand
|
|
173
|
+
* down so Ark decides. Known ids are unaffected. See
|
|
174
|
+
* {@link BytePlusVideoModelOrString} for how to discover and probe an id.
|
|
175
|
+
*
|
|
176
|
+
* @example
|
|
177
|
+
* ```typescript
|
|
178
|
+
* // Seedance 2.5, before this package ships probe-verified metadata for it:
|
|
179
|
+
* const adapter = byteplusVideo('dreamina-seedance-2-5-260628')
|
|
180
|
+
* ```
|
|
181
|
+
*/
|
|
182
|
+
export declare function byteplusVideo<TModel extends BytePlusVideoModelOrString>(model: TModel, config?: Omit<BytePlusVideoConfig, 'apiKey'>): BytePlusVideoAdapter<TModel>;
|