@tanstack/ai-byteplus 0.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +202 -0
  3. package/dist/esm/adapters/image.d.ts +89 -0
  4. package/dist/esm/adapters/image.js +229 -0
  5. package/dist/esm/adapters/image.js.map +1 -0
  6. package/dist/esm/adapters/text.d.ts +163 -0
  7. package/dist/esm/adapters/text.js +347 -0
  8. package/dist/esm/adapters/text.js.map +1 -0
  9. package/dist/esm/adapters/transcription.d.ts +102 -0
  10. package/dist/esm/adapters/transcription.js +274 -0
  11. package/dist/esm/adapters/transcription.js.map +1 -0
  12. package/dist/esm/adapters/tts.d.ts +143 -0
  13. package/dist/esm/adapters/tts.js +307 -0
  14. package/dist/esm/adapters/tts.js.map +1 -0
  15. package/dist/esm/adapters/video.d.ts +182 -0
  16. package/dist/esm/adapters/video.js +442 -0
  17. package/dist/esm/adapters/video.js.map +1 -0
  18. package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
  19. package/dist/esm/audio/tts-provider-options.d.ts +114 -0
  20. package/dist/esm/audio/wire-types.d.ts +261 -0
  21. package/dist/esm/audio/wire-types.js +28 -0
  22. package/dist/esm/audio/wire-types.js.map +1 -0
  23. package/dist/esm/image/image-provider-options.d.ts +165 -0
  24. package/dist/esm/image/image-provider-options.js +134 -0
  25. package/dist/esm/image/image-provider-options.js.map +1 -0
  26. package/dist/esm/image/wire-types.d.ts +149 -0
  27. package/dist/esm/index.d.ts +25 -0
  28. package/dist/esm/index.js +11 -0
  29. package/dist/esm/message-types.d.ts +154 -0
  30. package/dist/esm/model-meta.d.ts +594 -0
  31. package/dist/esm/model-meta.js +619 -0
  32. package/dist/esm/model-meta.js.map +1 -0
  33. package/dist/esm/text/text-provider-options.d.ts +109 -0
  34. package/dist/esm/utils/client.d.ts +183 -0
  35. package/dist/esm/utils/client.js +253 -0
  36. package/dist/esm/utils/client.js.map +1 -0
  37. package/dist/esm/video/video-provider-options.d.ts +197 -0
  38. package/dist/esm/video/video-provider-options.js +191 -0
  39. package/dist/esm/video/video-provider-options.js.map +1 -0
  40. package/dist/esm/video/wire-types.d.ts +248 -0
  41. package/package.json +77 -0
  42. package/src/adapters/image.ts +409 -0
  43. package/src/adapters/text.ts +539 -0
  44. package/src/adapters/transcription.ts +479 -0
  45. package/src/adapters/tts.ts +447 -0
  46. package/src/adapters/video.ts +732 -0
  47. package/src/audio/transcription-provider-options.ts +46 -0
  48. package/src/audio/tts-provider-options.ts +122 -0
  49. package/src/audio/wire-types.ts +290 -0
  50. package/src/image/image-provider-options.ts +288 -0
  51. package/src/image/wire-types.ts +169 -0
  52. package/src/index.ts +222 -0
  53. package/src/message-types.ts +169 -0
  54. package/src/model-meta.ts +954 -0
  55. package/src/text/text-provider-options.ts +151 -0
  56. package/src/utils/client.ts +377 -0
  57. package/src/video/video-provider-options.ts +361 -0
  58. package/src/video/wire-types.ts +293 -0
@@ -0,0 +1,447 @@
1
+ import { BaseTTSAdapter } from '@tanstack/ai/adapters'
2
+ import { generateId } from '@tanstack/ai-utils'
3
+ import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
4
+ import {
5
+ BYTEPLUS_VOICE_BASE_URL,
6
+ bytePlusVoiceError,
7
+ bytePlusVoiceHeaders,
8
+ getBytePlusVoiceApiKeyFromEnv,
9
+ readJsonBody,
10
+ withBytePlusVoiceDefaults,
11
+ } from '../utils/client'
12
+ import type { TTSOptions } from '@tanstack/ai'
13
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
14
+ import type { BytePlusVoiceConfig } from '../utils/client'
15
+ import type { BytePlusTTSModel } from '../model-meta'
16
+ import type {
17
+ BytePlusTTSAudioConfig,
18
+ BytePlusTTSAudioFormat,
19
+ BytePlusTTSCreateRequest,
20
+ BytePlusTTSCreateResponse,
21
+ } from '../audio/wire-types'
22
+ import type {
23
+ BytePlusTTSProviderOptions,
24
+ BytePlusTTSResult,
25
+ } from '../audio/tts-provider-options'
26
+
27
+ /** Path of the synchronous Seed Speech synthesis endpoint. */
28
+ const TTS_CREATE_PATH = '/api/v3/tts/create'
29
+
30
+ /**
31
+ * Name of the request field carrying the text to speak.
32
+ *
33
+ * **`text_prompt` is correct — do not "fix" this to `text`.** The endpoint
34
+ * schema (docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01) lists this
35
+ * body as exactly `model`, `text_prompt`, `references`, `audio_config`,
36
+ * `watermark`; there is no request-side `text`. The `text` spelling belongs
37
+ * to the *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits
38
+ * under `req_params.text`. The name stays isolated here so the two spellings
39
+ * never get conflated.
40
+ */
41
+ const TTS_TEXT_FIELD = 'text_prompt' satisfies keyof BytePlusTTSCreateRequest
42
+
43
+ /**
44
+ * Sample rate used when the caller doesn't pick one. The endpoint documents a
45
+ * default of 40000, which is not among the rates it accepts — so the adapter
46
+ * never relies on the server default and always sends this instead.
47
+ */
48
+ const DEFAULT_SAMPLE_RATE = 24000
49
+
50
+ /**
51
+ * True when a Seed Speech envelope's `code` means success.
52
+ *
53
+ * Seed Speech uses `0` for success and a flat numeric code otherwise
54
+ * (`45000010 Invalid X-Api-Key`). The success envelope has not been confirmed
55
+ * against a live key, so `code` is accepted as a number, its string form, or
56
+ * absent — an envelope that omits `code` entirely is treated as success, which
57
+ * is what the HTTP status already told us.
58
+ */
59
+ function isZeroCode(code: number | string | undefined): boolean {
60
+ if (code === undefined) return true
61
+ return Number(code) === 0
62
+ }
63
+
64
+ /**
65
+ * Voice used when neither `TTSOptions.voice` nor `modelOptions.speaker` is
66
+ * set — the English female "Stokie" voice from the TTS 2.0 generation.
67
+ */
68
+ export const BYTEPLUS_DEFAULT_TTS_SPEAKER = 'en_female_stokie_uranus_bigtts'
69
+
70
+ /**
71
+ * Hard cap on a single Seed Speech synthesis, in seconds. Longer scripts must
72
+ * be split across calls and stitched client-side.
73
+ *
74
+ * The cap applies to the *pre-rate* length the service bills on — the
75
+ * `original_duration` it returns. The delivered clip can run longer than this
76
+ * when `speech_rate` slows it down.
77
+ */
78
+ export const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120
79
+
80
+ /**
81
+ * BytePlus Seed Speech text-to-speech adapter.
82
+ *
83
+ * Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.
84
+ * Two things differ from the Ark-hosted adapters in this package:
85
+ *
86
+ * - **Separate API key.** Seed Speech authenticates with `X-Api-Key` and its
87
+ * own key (`BYTEPLUS_VOICE_API_KEY`). An `ARK_API_KEY` sent here is
88
+ * rejected with `45000010 Invalid X-Api-Key`.
89
+ * - **120 s output cap.** A single call synthesises at most
90
+ * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} seconds of audio; longer scripts
91
+ * have to be split and stitched client-side. The cap is measured before
92
+ * `speech_rate` is applied, so a slowed clip can play for longer than that
93
+ * — `result.duration` is the delivered length and `result.originalDuration`
94
+ * is the metered one.
95
+ *
96
+ * Streaming synthesis (`tts/unidirectional` and the WebSocket API) is not
97
+ * covered by this adapter — note that endpoint spells the text field `text`,
98
+ * while this one uses `text_prompt`.
99
+ *
100
+ * @example
101
+ * ```ts
102
+ * const adapter = byteplusSpeech('seed-audio-1.0')
103
+ * const result = await generateSpeech({
104
+ * adapter,
105
+ * text: 'welcome to the guitar store',
106
+ * voice: 'en_female_stokie_uranus_bigtts',
107
+ * format: 'mp3',
108
+ * })
109
+ * ```
110
+ */
111
+ export class BytePlusTTSAdapter<
112
+ TModel extends BytePlusTTSModel = BytePlusTTSModel,
113
+ > extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {
114
+ readonly name = 'byteplus' as const
115
+
116
+ private readonly apiKey: string
117
+ private readonly baseURL: string
118
+ private readonly defaultHeaders: Record<string, string>
119
+ private readonly fetchImpl: typeof fetch
120
+
121
+ constructor(model: TModel, config: BytePlusVoiceConfig) {
122
+ super(model, config)
123
+ const resolved = withBytePlusVoiceDefaults(config)
124
+ this.apiKey = resolved.apiKey
125
+ this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL
126
+ this.defaultHeaders = resolved.defaultHeaders ?? {}
127
+ this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)
128
+ }
129
+
130
+ async generateSpeech(
131
+ options: TTSOptions<BytePlusTTSProviderOptions>,
132
+ ): Promise<BytePlusTTSResult> {
133
+ const { logger, model, text, voice, format, speed, modelOptions } = options
134
+
135
+ logger.request(`activity=generateSpeech provider=byteplus model=${model}`, {
136
+ provider: 'byteplus',
137
+ model,
138
+ })
139
+
140
+ const { body, audioFormat, sampleRate } = buildTTSRequestBody({
141
+ model,
142
+ text,
143
+ voice,
144
+ format,
145
+ speed,
146
+ modelOptions,
147
+ logger,
148
+ })
149
+
150
+ try {
151
+ const response = await this.fetchImpl(
152
+ `${this.baseURL}${TTS_CREATE_PATH}`,
153
+ {
154
+ method: 'POST',
155
+ headers: bytePlusVoiceHeaders(this.apiKey, {
156
+ ...this.defaultHeaders,
157
+ // Client-generated per-request id. BytePlus echoes it in their
158
+ // request logs, which is what support asks for when diagnosing a
159
+ // synthesis failure.
160
+ 'X-Api-Request-Id': newRequestId(),
161
+ }),
162
+ body: JSON.stringify(body),
163
+ },
164
+ )
165
+
166
+ const payload = await readJsonBody(response)
167
+
168
+ if (!response.ok) {
169
+ throw bytePlusVoiceError(response.status, payload, 'text-to-speech')
170
+ }
171
+
172
+ const data = payload as BytePlusTTSCreateResponse
173
+
174
+ // Seed Speech reports status in the body, not only in the HTTP status:
175
+ // a 200 can carry a non-zero `code`. Check it before looking at `audio`,
176
+ // because a failed call may still return a partial or placeholder
177
+ // payload that would otherwise be handed back as if it were valid.
178
+ //
179
+ // `code` is accepted as a number *or* a string. The success envelope was
180
+ // never confirmed against a live key (no voice key yet — see
181
+ // `audio/wire-types.ts`), and `readStringField` already tolerates both
182
+ // forms when rendering the error, so requiring a number here would let
183
+ // `{"code": "45000010"}` through both this gate and the one below.
184
+ if (!isZeroCode(data.code)) {
185
+ throw bytePlusVoiceError(response.status, payload, 'text-to-speech')
186
+ }
187
+
188
+ // Belt and braces for a 200 that reports success but carries nothing to
189
+ // play. Say that the adapter rejected it, rather than reusing the
190
+ // envelope-error phrasing — a bare "failed (200)" gives no hint that the
191
+ // response was well-formed and simply empty.
192
+ if (typeof data.audio !== 'string' || data.audio.length === 0) {
193
+ throw new Error(
194
+ `BytePlus Seed Speech text-to-speech returned a success response ` +
195
+ `with no audio (model ${model}).`,
196
+ )
197
+ }
198
+
199
+ const duration = toDurationSeconds(data.duration)
200
+ const originalDuration = toDurationSeconds(data.original_duration)
201
+
202
+ return {
203
+ id: generateId(this.name),
204
+ model,
205
+ audio: data.audio,
206
+ format: audioFormat,
207
+ contentType: getContentType(audioFormat, sampleRate),
208
+ ...(duration !== undefined && { duration }),
209
+ ...(originalDuration !== undefined && { originalDuration }),
210
+ ...(data.subtitle !== undefined && { subtitle: data.subtitle }),
211
+ ...(data.url !== undefined && { url: data.url }),
212
+ }
213
+ } catch (error) {
214
+ logger.errors('byteplus.generateSpeech fatal', {
215
+ error: toRunErrorPayload(error, 'byteplus.generateSpeech failed'),
216
+ source: 'byteplus.generateSpeech',
217
+ })
218
+ throw error
219
+ }
220
+ }
221
+ }
222
+
223
+ /**
224
+ * Build the JSON body for `POST /api/v3/tts/create`, resolving the voice,
225
+ * output format and rate fields in one place.
226
+ *
227
+ * Returns the request `body`, the resolved `audioFormat` and the
228
+ * `sampleRate`, which the caller reports on the result and turns into a
229
+ * `contentType`.
230
+ */
231
+ export function buildTTSRequestBody(options: {
232
+ model: string
233
+ text: string
234
+ voice: string | undefined
235
+ format: TTSOptions['format'] | undefined
236
+ speed: number | undefined
237
+ modelOptions: BytePlusTTSProviderOptions | undefined
238
+ logger: InternalLogger
239
+ }): {
240
+ body: BytePlusTTSCreateRequest
241
+ audioFormat: BytePlusTTSAudioFormat
242
+ sampleRate: number
243
+ } {
244
+ const { model, text, voice, format, speed, modelOptions, logger } = options
245
+
246
+ const audioFormat = pickAudioFormat(modelOptions?.format, format, logger)
247
+ // Always explicit: the documented server default (40000) is not one of the
248
+ // rates the endpoint accepts, so relying on it is a coin flip.
249
+ const sampleRate = modelOptions?.sample_rate ?? DEFAULT_SAMPLE_RATE
250
+
251
+ const audioConfig: BytePlusTTSAudioConfig = {
252
+ format: audioFormat,
253
+ sample_rate: sampleRate,
254
+ }
255
+ if (modelOptions?.pitch_rate !== undefined) {
256
+ audioConfig.pitch_rate = modelOptions.pitch_rate
257
+ }
258
+ if (modelOptions?.loudness_rate !== undefined) {
259
+ audioConfig.loudness_rate = modelOptions.loudness_rate
260
+ }
261
+ if (modelOptions?.enable_subtitle !== undefined) {
262
+ audioConfig.enable_subtitle = modelOptions.enable_subtitle
263
+ }
264
+
265
+ // An explicit `speech_rate` always wins over the derived one — it is the
266
+ // native unit and the only way to reach the extremes precisely.
267
+ const speechRate =
268
+ modelOptions?.speech_rate ??
269
+ (speed !== undefined ? toSpeechRate(speed, logger) : undefined)
270
+ if (speechRate !== undefined) {
271
+ audioConfig.speech_rate = speechRate
272
+ }
273
+
274
+ const body: BytePlusTTSCreateRequest = {
275
+ model,
276
+ [TTS_TEXT_FIELD]: text,
277
+ // The voice belongs inside `references`, not at the top level — a
278
+ // top-level `speaker` is silently ignored by the server. The flat member
279
+ // shape here is the best-supported reading of the docs; see
280
+ // `BytePlusTTSReference` for the unresolved part and the live-probe flag.
281
+ references: modelOptions?.references ?? [
282
+ {
283
+ speaker: modelOptions?.speaker ?? voice ?? BYTEPLUS_DEFAULT_TTS_SPEAKER,
284
+ },
285
+ ],
286
+ audio_config: audioConfig,
287
+ }
288
+ if (modelOptions?.watermark !== undefined) {
289
+ body.watermark = modelOptions.watermark
290
+ }
291
+
292
+ return { body, audioFormat, sampleRate }
293
+ }
294
+
295
+ /**
296
+ * Convert the cross-provider `TTSOptions.speed` multiplier into Seed Speech's
297
+ * `speech_rate` percentage.
298
+ *
299
+ * ```
300
+ * speech_rate = clamp(round((speed - 1) * 100), -50, 100)
301
+ * ```
302
+ *
303
+ * so `0.5 → -50`, `1.0 → 0`, `1.5 → 50`, `2.0 → 100`.
304
+ *
305
+ * The range `-50`..`100` and both multiplier anchors (`-50` = 0.5×,
306
+ * `100` = 2×) are documented, and hold on both the `/tts/create` and
307
+ * `/tts/unidirectional` endpoints.
308
+ *
309
+ * `TTSOptions.speed` spans a wider 0.25×–4× than the endpoint supports, so
310
+ * anything outside 0.5×–2× clamps (and warns) rather than erroring.
311
+ */
312
+ export function toSpeechRate(speed: number, logger?: InternalLogger): number {
313
+ const rate = Math.round((speed - 1) * 100)
314
+ const clamped = Math.min(100, Math.max(-50, rate))
315
+ if (clamped !== rate) {
316
+ logger?.warn(
317
+ `Speed ${speed}× is outside the range BytePlus Seed Speech documents (0.5×–2×) — clamping speech_rate from ${rate} to ${clamped}.`,
318
+ { provider: 'byteplus', requestedSpeed: speed, speechRate: clamped },
319
+ )
320
+ }
321
+ return clamped
322
+ }
323
+
324
+ /**
325
+ * Map the cross-provider `TTSOptions.format` onto a Seed Speech output
326
+ * format. An explicit `modelOptions.format` always wins.
327
+ *
328
+ * Seed Speech produces `wav`, `mp3`, `pcm` and `ogg_opus`. The generic
329
+ * `opus` maps onto `ogg_opus`; `aac` and `flac` have no equivalent and fall
330
+ * back to `mp3` (matching how the other non-OpenAI TTS adapters in this repo
331
+ * handle unsupported codecs). The fallback is logged so it isn't silent.
332
+ */
333
+ function pickAudioFormat(
334
+ override: BytePlusTTSAudioFormat | undefined,
335
+ format: TTSOptions['format'] | undefined,
336
+ logger: InternalLogger,
337
+ ): BytePlusTTSAudioFormat {
338
+ if (override) return override
339
+ if (!format) return 'mp3'
340
+ switch (format) {
341
+ case 'mp3':
342
+ case 'wav':
343
+ case 'pcm':
344
+ return format
345
+ case 'opus':
346
+ return 'ogg_opus'
347
+ case 'aac':
348
+ case 'flac':
349
+ logger.warn(
350
+ `BytePlus Seed Speech does not support ${format} output — falling back to mp3. Set modelOptions.format to choose between wav, mp3, pcm and ogg_opus.`,
351
+ { provider: 'byteplus', requestedFormat: format },
352
+ )
353
+ return 'mp3'
354
+ }
355
+ }
356
+
357
+ /**
358
+ * Build a per-request id for the `X-Api-Request-Id` header. Falls back to the
359
+ * package's own id generator on runtimes without `crypto.randomUUID`.
360
+ */
361
+ function newRequestId(): string {
362
+ return globalThis.crypto?.randomUUID?.() ?? generateId('byteplus-tts')
363
+ }
364
+
365
+ /**
366
+ * MIME type for a Seed Speech output format.
367
+ *
368
+ * `pcm` is raw little-endian 16-bit samples, so its media type has to carry
369
+ * the sample rate (RFC 3551/3555). The adapter always resolves a rate, so one
370
+ * is always available here.
371
+ */
372
+ export function getContentType(
373
+ format: BytePlusTTSAudioFormat,
374
+ sampleRate?: number,
375
+ ): string {
376
+ switch (format) {
377
+ case 'mp3':
378
+ return 'audio/mpeg'
379
+ case 'wav':
380
+ return 'audio/wav'
381
+ case 'ogg_opus':
382
+ return 'audio/ogg;codecs=opus'
383
+ case 'pcm':
384
+ return `audio/L16;rate=${sampleRate ?? 24000}`
385
+ }
386
+ }
387
+
388
+ /**
389
+ * Coerce a `duration` / `original_duration` field to a usable number of
390
+ * seconds.
391
+ *
392
+ * Both are documented as float **seconds**, so this only parses the string
393
+ * form and drops values that can't be a length (zero, negative, non-numeric).
394
+ * Note that `duration` may legitimately exceed
395
+ * {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} — a clip synthesised at
396
+ * `speech_rate: -50` runs at half speed, so up to 120 s of billed audio can
397
+ * be delivered as up to 240 s of playback. Do not "correct" a large value
398
+ * here; the cap applies to `original_duration`.
399
+ *
400
+ * The subtitle timings are the mixed-unit exception: those are milliseconds
401
+ * and are passed through untouched on {@link BytePlusTTSResult.subtitle}.
402
+ */
403
+ export function toDurationSeconds(
404
+ raw: number | string | undefined,
405
+ ): number | undefined {
406
+ const value = typeof raw === 'string' ? Number(raw) : raw
407
+ if (value === undefined || !Number.isFinite(value) || value <= 0) {
408
+ return undefined
409
+ }
410
+ return value
411
+ }
412
+
413
+ /**
414
+ * Creates a BytePlus Seed Speech TTS adapter with an explicit API key.
415
+ *
416
+ * The key is the **Seed Speech** key, not the Ark key used by the chat, image
417
+ * and video adapters.
418
+ *
419
+ * @example
420
+ * ```ts
421
+ * const adapter = createBytePlusSpeech('seed-audio-1.0', process.env.BYTEPLUS_VOICE_API_KEY!)
422
+ * ```
423
+ */
424
+ export function createBytePlusSpeech<
425
+ TModel extends BytePlusTTSModel = BytePlusTTSModel,
426
+ >(
427
+ model: TModel,
428
+ apiKey: string,
429
+ config?: Omit<BytePlusVoiceConfig, 'apiKey'>,
430
+ ): BytePlusTTSAdapter<TModel> {
431
+ return new BytePlusTTSAdapter(model, { ...config, apiKey })
432
+ }
433
+
434
+ /**
435
+ * Creates a BytePlus Seed Speech TTS adapter, reading the API key from
436
+ * `BYTEPLUS_VOICE_API_KEY`.
437
+ *
438
+ * @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.
439
+ */
440
+ export function byteplusSpeech<
441
+ TModel extends BytePlusTTSModel = BytePlusTTSModel,
442
+ >(
443
+ model: TModel,
444
+ config?: Omit<BytePlusVoiceConfig, 'apiKey'>,
445
+ ): BytePlusTTSAdapter<TModel> {
446
+ return createBytePlusSpeech(model, getBytePlusVoiceApiKeyFromEnv(), config)
447
+ }