@tanstack/ai-byteplus 0.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +202 -0
  3. package/dist/esm/adapters/image.d.ts +89 -0
  4. package/dist/esm/adapters/image.js +229 -0
  5. package/dist/esm/adapters/image.js.map +1 -0
  6. package/dist/esm/adapters/text.d.ts +163 -0
  7. package/dist/esm/adapters/text.js +347 -0
  8. package/dist/esm/adapters/text.js.map +1 -0
  9. package/dist/esm/adapters/transcription.d.ts +102 -0
  10. package/dist/esm/adapters/transcription.js +274 -0
  11. package/dist/esm/adapters/transcription.js.map +1 -0
  12. package/dist/esm/adapters/tts.d.ts +143 -0
  13. package/dist/esm/adapters/tts.js +307 -0
  14. package/dist/esm/adapters/tts.js.map +1 -0
  15. package/dist/esm/adapters/video.d.ts +182 -0
  16. package/dist/esm/adapters/video.js +442 -0
  17. package/dist/esm/adapters/video.js.map +1 -0
  18. package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
  19. package/dist/esm/audio/tts-provider-options.d.ts +114 -0
  20. package/dist/esm/audio/wire-types.d.ts +261 -0
  21. package/dist/esm/audio/wire-types.js +28 -0
  22. package/dist/esm/audio/wire-types.js.map +1 -0
  23. package/dist/esm/image/image-provider-options.d.ts +165 -0
  24. package/dist/esm/image/image-provider-options.js +134 -0
  25. package/dist/esm/image/image-provider-options.js.map +1 -0
  26. package/dist/esm/image/wire-types.d.ts +149 -0
  27. package/dist/esm/index.d.ts +25 -0
  28. package/dist/esm/index.js +11 -0
  29. package/dist/esm/message-types.d.ts +154 -0
  30. package/dist/esm/model-meta.d.ts +594 -0
  31. package/dist/esm/model-meta.js +619 -0
  32. package/dist/esm/model-meta.js.map +1 -0
  33. package/dist/esm/text/text-provider-options.d.ts +109 -0
  34. package/dist/esm/utils/client.d.ts +183 -0
  35. package/dist/esm/utils/client.js +253 -0
  36. package/dist/esm/utils/client.js.map +1 -0
  37. package/dist/esm/video/video-provider-options.d.ts +197 -0
  38. package/dist/esm/video/video-provider-options.js +191 -0
  39. package/dist/esm/video/video-provider-options.js.map +1 -0
  40. package/dist/esm/video/wire-types.d.ts +248 -0
  41. package/package.json +77 -0
  42. package/src/adapters/image.ts +409 -0
  43. package/src/adapters/text.ts +539 -0
  44. package/src/adapters/transcription.ts +479 -0
  45. package/src/adapters/tts.ts +447 -0
  46. package/src/adapters/video.ts +732 -0
  47. package/src/audio/transcription-provider-options.ts +46 -0
  48. package/src/audio/tts-provider-options.ts +122 -0
  49. package/src/audio/wire-types.ts +290 -0
  50. package/src/image/image-provider-options.ts +288 -0
  51. package/src/image/wire-types.ts +169 -0
  52. package/src/index.ts +222 -0
  53. package/src/message-types.ts +169 -0
  54. package/src/model-meta.ts +954 -0
  55. package/src/text/text-provider-options.ts +151 -0
  56. package/src/utils/client.ts +377 -0
  57. package/src/video/video-provider-options.ts +361 -0
  58. package/src/video/wire-types.ts +293 -0
@@ -0,0 +1,479 @@
1
+ import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
2
+ import { arrayBufferToBase64, generateId } from '@tanstack/ai-utils'
3
+ import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
4
+ import {
5
+ BYTEPLUS_VOICE_BASE_URL,
6
+ bytePlusVoiceError,
7
+ bytePlusVoiceHeaders,
8
+ getBytePlusVoiceApiKeyFromEnv,
9
+ readJsonBody,
10
+ withBytePlusVoiceDefaults,
11
+ } from '../utils/client'
12
+ import {
13
+ BYTEPLUS_ASR_RESOURCE_HEADER,
14
+ BYTEPLUS_ASR_RESOURCE_ID,
15
+ } from '../audio/wire-types'
16
+ import type {
17
+ TokenUsage,
18
+ TranscriptionOptions,
19
+ TranscriptionResult,
20
+ TranscriptionSegment,
21
+ TranscriptionWord,
22
+ } from '@tanstack/ai'
23
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
24
+ import type { BytePlusVoiceConfig } from '../utils/client'
25
+ import type { BytePlusTranscriptionModel } from '../model-meta'
26
+ import type {
27
+ BytePlusASRAudio,
28
+ BytePlusASRRecognizeRequest,
29
+ BytePlusASRRecognizeResponse,
30
+ BytePlusASRUtterance,
31
+ } from '../audio/wire-types'
32
+ import type { BytePlusTranscriptionProviderOptions } from '../audio/transcription-provider-options'
33
+
34
+ /** Path of the synchronous ("flash") Seed ASR endpoint. */
35
+ const RECOGNIZE_FLASH_PATH = '/api/v3/auc/bigmodel/recognize/flash'
36
+
37
+ /**
38
+ * BytePlus-specific extension of `TranscriptionWord` carrying the per-word
39
+ * confidence Seed ASR returns. The cross-provider contract has no field for
40
+ * it, so callers who want it narrow the array — the same pattern the Grok
41
+ * adapter uses:
42
+ *
43
+ * ```ts
44
+ * const words = result.words as Array<BytePlusTranscriptionWord> | undefined
45
+ * ```
46
+ */
47
+ export interface BytePlusTranscriptionWord extends TranscriptionWord {
48
+ /** Model confidence for the word, when Seed ASR returns one. */
49
+ confidence?: number
50
+ }
51
+
52
+ /** Default `user.uid` echoed into BytePlus' request logs. */
53
+ const DEFAULT_UID = 'tanstack-ai'
54
+
55
+ /**
56
+ * BytePlus Seed Speech transcription (ASR) adapter.
57
+ *
58
+ * Talks to `POST {baseURL}/api/v3/auc/bigmodel/recognize/flash` — the
59
+ * synchronous "flash" endpoint, which returns the whole transcript in one
60
+ * response rather than requiring a submit/poll cycle. It accepts audio up to
61
+ * 2 hours long or 100 MB, either as a publicly reachable URL or as base64
62
+ * bytes.
63
+ *
64
+ * Two BytePlus-specific details:
65
+ *
66
+ * - The model is selected by the `X-Api-Resource-Id` header
67
+ * (`volc.seedasr.auc_turbo`), not by a `model` field in the body. The
68
+ * package's `seed-asr` model id exists to satisfy the SDK contract and to
69
+ * give logs a stable value.
70
+ * - Authentication uses `X-Api-Key` with the **Seed Speech** key, which is a
71
+ * different key from `ARK_API_KEY`.
72
+ *
73
+ * All timings on the wire are milliseconds; they are converted to seconds to
74
+ * match the cross-provider `TranscriptionResult`.
75
+ *
76
+ * @example
77
+ * ```ts
78
+ * const adapter = byteplusTranscription('seed-asr')
79
+ * const result = await generateTranscription({
80
+ * adapter,
81
+ * audio: 'https://example.com/interview.mp3',
82
+ * language: 'en-US',
83
+ * })
84
+ * ```
85
+ */
86
+ export class BytePlusTranscriptionAdapter<
87
+ TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,
88
+ > extends BaseTranscriptionAdapter<
89
+ TModel,
90
+ BytePlusTranscriptionProviderOptions
91
+ > {
92
+ readonly name = 'byteplus' as const
93
+
94
+ private readonly apiKey: string
95
+ private readonly baseURL: string
96
+ private readonly defaultHeaders: Record<string, string>
97
+ private readonly fetchImpl: typeof fetch
98
+
99
+ constructor(model: TModel, config: BytePlusVoiceConfig) {
100
+ super(model, config)
101
+ const resolved = withBytePlusVoiceDefaults(config)
102
+ this.apiKey = resolved.apiKey
103
+ this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL
104
+ this.defaultHeaders = resolved.defaultHeaders ?? {}
105
+ this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)
106
+ }
107
+
108
+ async transcribe(
109
+ options: TranscriptionOptions<BytePlusTranscriptionProviderOptions>,
110
+ ): Promise<TranscriptionResult> {
111
+ const {
112
+ logger,
113
+ model,
114
+ audio,
115
+ language,
116
+ prompt,
117
+ responseFormat,
118
+ modelOptions,
119
+ } = options
120
+
121
+ logger.request(
122
+ `activity=generateTranscription provider=byteplus model=${model}`,
123
+ { provider: 'byteplus', model },
124
+ )
125
+
126
+ if (prompt) {
127
+ logger.warn(
128
+ 'BytePlus Seed ASR has no prompt-biasing field on the flash endpoint — the `prompt` option is ignored.',
129
+ { provider: 'byteplus', model },
130
+ )
131
+ }
132
+
133
+ // The flash endpoint answers with one JSON shape and offers no format
134
+ // negotiation, so srt/vtt/text/verbose_json can't be honoured. `segments`
135
+ // on the result carry the timings a caller would have wanted from srt/vtt.
136
+ if (responseFormat !== undefined && responseFormat !== 'json') {
137
+ logger.warn(
138
+ `BytePlus Seed ASR always returns JSON — the requested responseFormat "${responseFormat}" is ignored. Build srt/vtt from result.segments if you need them.`,
139
+ { provider: 'byteplus', model, responseFormat },
140
+ )
141
+ }
142
+
143
+ try {
144
+ const audioPayload = await normalizeAudioInput(
145
+ audio,
146
+ modelOptions?.audio_format,
147
+ )
148
+ const body = buildRecognizeRequestBody({
149
+ audio: audioPayload,
150
+ language,
151
+ modelOptions,
152
+ })
153
+
154
+ const response = await this.fetchImpl(
155
+ `${this.baseURL}${RECOGNIZE_FLASH_PATH}`,
156
+ {
157
+ method: 'POST',
158
+ headers: bytePlusVoiceHeaders(this.apiKey, {
159
+ ...this.defaultHeaders,
160
+ [BYTEPLUS_ASR_RESOURCE_HEADER]: BYTEPLUS_ASR_RESOURCE_ID,
161
+ }),
162
+ body: JSON.stringify(body),
163
+ },
164
+ )
165
+
166
+ const payload = await readJsonBody(response)
167
+
168
+ if (!response.ok) {
169
+ throw bytePlusVoiceError(response.status, payload, 'transcription')
170
+ }
171
+
172
+ const data = payload as BytePlusASRRecognizeResponse
173
+ const text = data.result?.text ?? data.transcript
174
+
175
+ // The flash endpoint can answer HTTP 200 while carrying the numeric
176
+ // error envelope, so an absent transcript is a failure rather than an
177
+ // empty result.
178
+ if (typeof text !== 'string') {
179
+ throw bytePlusVoiceError(response.status, payload, 'transcription')
180
+ }
181
+
182
+ // An empty string is well-formed, so it isn't an error — silence is a
183
+ // legitimate transcription. But it is also what a 200-wrapped failure
184
+ // looks like, so say so rather than handing back a successful, empty
185
+ // result with no signal.
186
+ if (text === '' && !hasUtterances(data)) {
187
+ logger.warn(
188
+ `byteplus: transcription returned an empty transcript with no ` +
189
+ `utterances. This is a valid result for silent audio, and is also ` +
190
+ `what a 200-wrapped failure looks like.`,
191
+ { provider: this.name, model },
192
+ )
193
+ }
194
+
195
+ // Seed ASR doesn't echo the language back, so report the one that was
196
+ // actually sent — which is `modelOptions.language` when it overrode the
197
+ // cross-provider hint.
198
+ const requestedLanguage = modelOptions?.language ?? language
199
+
200
+ return {
201
+ id: generateId(this.name),
202
+ model,
203
+ ...mapRecognizeResponse(data, text, logger),
204
+ ...(requestedLanguage !== undefined && { language: requestedLanguage }),
205
+ }
206
+ } catch (error) {
207
+ logger.errors('byteplus.transcribe fatal', {
208
+ error: toRunErrorPayload(error, 'byteplus.transcribe failed'),
209
+ source: 'byteplus.transcribe',
210
+ })
211
+ throw error
212
+ }
213
+ }
214
+ }
215
+
216
+ /**
217
+ * Build the JSON body for `POST /api/v3/auc/bigmodel/recognize/flash`.
218
+ *
219
+ * `show_utterances` defaults to `true` so the response carries the
220
+ * per-utterance breakdown that populates `segments` and `words`.
221
+ */
222
+ export function buildRecognizeRequestBody(options: {
223
+ audio: BytePlusASRAudio
224
+ language: string | undefined
225
+ modelOptions: BytePlusTranscriptionProviderOptions | undefined
226
+ }): BytePlusASRRecognizeRequest {
227
+ const { audio, language, modelOptions } = options
228
+
229
+ const resolvedLanguage = modelOptions?.language ?? language
230
+
231
+ return {
232
+ user: { uid: modelOptions?.uid ?? DEFAULT_UID },
233
+ audio,
234
+ request: {
235
+ model_name: modelOptions?.model_name ?? 'bigmodel',
236
+ show_utterances: modelOptions?.show_utterances ?? true,
237
+ ...(modelOptions?.enable_itn !== undefined && {
238
+ enable_itn: modelOptions.enable_itn,
239
+ }),
240
+ ...(modelOptions?.enable_punc !== undefined && {
241
+ enable_punc: modelOptions.enable_punc,
242
+ }),
243
+ ...(modelOptions?.enable_ddc !== undefined && {
244
+ enable_ddc: modelOptions.enable_ddc,
245
+ }),
246
+ ...(modelOptions?.enable_speaker_info !== undefined && {
247
+ enable_speaker_info: modelOptions.enable_speaker_info,
248
+ }),
249
+ ...(resolvedLanguage !== undefined && { language: resolvedLanguage }),
250
+ },
251
+ }
252
+ }
253
+
254
+ /**
255
+ * Turn a recognition response into the transcript-shaped half of a
256
+ * `TranscriptionResult`. Wire timings are milliseconds; everything returned
257
+ * here is seconds.
258
+ */
259
+ export function mapRecognizeResponse(
260
+ data: BytePlusASRRecognizeResponse,
261
+ text: string,
262
+ logger?: InternalLogger,
263
+ ): Omit<TranscriptionResult, 'id' | 'model'> {
264
+ const utterances = data.result?.utterances ?? data.utterances ?? []
265
+ // `id` numbers the segments we emit, not the utterances we were given, so
266
+ // dropping an untimed utterance doesn't leave a hole in the sequence.
267
+ const segments = utterances
268
+ .flatMap((utterance) => toSegment(utterance))
269
+ .map((segment, index) => ({ ...segment, id: index }))
270
+
271
+ const rawWords = utterances.flatMap((utterance) => utterance.words ?? [])
272
+ const words = rawWords.flatMap((word) => {
273
+ if (
274
+ typeof word.text !== 'string' ||
275
+ typeof word.start_time !== 'number' ||
276
+ typeof word.end_time !== 'number'
277
+ ) {
278
+ return []
279
+ }
280
+ const mapped: BytePlusTranscriptionWord = {
281
+ word: word.text,
282
+ start: msToSeconds(word.start_time),
283
+ end: msToSeconds(word.end_time),
284
+ }
285
+ if (word.confidence !== undefined) mapped.confidence = word.confidence
286
+ return [mapped]
287
+ })
288
+
289
+ // Untimed entries are dropped rather than emitted with NaN timings, but a
290
+ // silent drop leaves the caller unable to tell "the provider sent no
291
+ // timings" from "the adapter discarded them" — the two have very different
292
+ // fixes, and a field rename upstream (e.g. `text` → `word`) would empty
293
+ // these arrays without a single error anywhere.
294
+ const droppedWords = rawWords.length - words.length
295
+ if (droppedWords > 0) {
296
+ logger?.warn(
297
+ `byteplus: dropped ${droppedWords} of ${rawWords.length} word(s) with ` +
298
+ `missing or non-numeric timings.`,
299
+ { provider: 'byteplus' },
300
+ )
301
+ }
302
+ const droppedSegments = utterances.length - segments.length
303
+ if (droppedSegments > 0) {
304
+ logger?.warn(
305
+ `byteplus: dropped ${droppedSegments} of ${utterances.length} ` +
306
+ `utterance(s) with missing or non-numeric timings.`,
307
+ { provider: 'byteplus' },
308
+ )
309
+ }
310
+
311
+ const durationMs = data.audio_info?.duration
312
+ const duration =
313
+ typeof durationMs === 'number' && durationMs > 0
314
+ ? msToSeconds(durationMs)
315
+ : undefined
316
+
317
+ // Seed ASR is duration-billed and reports no token counts, so `usage`
318
+ // carries only the audio length — the same shape the Grok and OpenAI
319
+ // whisper paths use.
320
+ const usage: TokenUsage | undefined =
321
+ duration !== undefined
322
+ ? {
323
+ promptTokens: 0,
324
+ completionTokens: 0,
325
+ totalTokens: 0,
326
+ durationSeconds: duration,
327
+ }
328
+ : undefined
329
+
330
+ return {
331
+ text,
332
+ ...(duration !== undefined && { duration }),
333
+ ...(segments.length > 0 && { segments }),
334
+ ...(words.length > 0 && { words }),
335
+ ...(usage !== undefined && { usage }),
336
+ }
337
+ }
338
+
339
+ /**
340
+ * Convert one utterance into a segment, or nothing when it carries no
341
+ * timings. The `id` is a placeholder — the caller renumbers after filtering.
342
+ */
343
+ /**
344
+ * True when the response carries at least one utterance, in either envelope
345
+ * form. Used to tell "silent audio" from a 200-wrapped failure: a genuinely
346
+ * empty transcript usually still arrives with no utterances, so the pairing is
347
+ * a hint rather than proof — hence a warning rather than a throw.
348
+ */
349
+ function hasUtterances(data: BytePlusASRRecognizeResponse): boolean {
350
+ return (data.result?.utterances ?? data.utterances ?? []).length > 0
351
+ }
352
+
353
+ function toSegment(
354
+ utterance: BytePlusASRUtterance,
355
+ ): Array<TranscriptionSegment> {
356
+ if (
357
+ typeof utterance.start_time !== 'number' ||
358
+ typeof utterance.end_time !== 'number'
359
+ ) {
360
+ return []
361
+ }
362
+ const speaker = utterance.additions?.speaker
363
+ return [
364
+ {
365
+ id: 0,
366
+ start: msToSeconds(utterance.start_time),
367
+ end: msToSeconds(utterance.end_time),
368
+ text: utterance.text ?? '',
369
+ ...(speaker !== undefined && { speaker }),
370
+ },
371
+ ]
372
+ }
373
+
374
+ /**
375
+ * **Must verify when the Seed Speech key lands.** Every timing this adapter
376
+ * reads — `audio_info.duration`, and each utterance's and word's
377
+ * `start_time` / `end_time` — is assumed to be milliseconds. That comes from
378
+ * the Volcengine flash-recognition reference this endpoint derives from
379
+ * (a 2.499 s clip reports `duration: 2499`), not from a BytePlus response we
380
+ * have seen. If BytePlus reports seconds instead, every duration, segment and
381
+ * word timing here is 1000× too small, and this is the only place to fix.
382
+ */
383
+ function msToSeconds(milliseconds: number): number {
384
+ return milliseconds / 1000
385
+ }
386
+
387
+ /**
388
+ * Turn the cross-provider `audio` input into the endpoint's `audio` block.
389
+ *
390
+ * URLs are passed through untouched — Seed ASR fetches them itself, which
391
+ * avoids pulling large media through this process. Everything else is sent as
392
+ * base64 `data`, with the container inferred from the input's MIME type or
393
+ * filename when the caller didn't pin `audio_format`.
394
+ */
395
+ export async function normalizeAudioInput(
396
+ audio: TranscriptionOptions['audio'],
397
+ formatHint: string | undefined,
398
+ ): Promise<BytePlusASRAudio> {
399
+ const withFormat = (
400
+ payload: BytePlusASRAudio,
401
+ inferred?: string,
402
+ ): BytePlusASRAudio => {
403
+ const format = formatHint ?? inferred
404
+ return format ? { ...payload, format } : payload
405
+ }
406
+
407
+ if (typeof audio === 'string') {
408
+ if (/^https?:\/\//i.test(audio)) {
409
+ return withFormat({ url: audio }, extensionOf(audio))
410
+ }
411
+ const dataUrl = /^data:([^;,]+)?(?:;[^,]*)*,(.*)$/s.exec(audio)
412
+ if (dataUrl) {
413
+ return withFormat({ data: dataUrl[2] ?? '' }, formatFromMime(dataUrl[1]))
414
+ }
415
+ // A bare string that is neither a URL nor a data URL is already base64.
416
+ return withFormat({ data: audio })
417
+ }
418
+
419
+ if (audio instanceof ArrayBuffer) {
420
+ return withFormat({ data: arrayBufferToBase64(audio) })
421
+ }
422
+
423
+ const data = arrayBufferToBase64(await audio.arrayBuffer())
424
+ const inferred =
425
+ ('name' in audio && typeof audio.name === 'string'
426
+ ? extensionOf(audio.name)
427
+ : undefined) ?? formatFromMime(audio.type)
428
+ return withFormat({ data }, inferred)
429
+ }
430
+
431
+ function extensionOf(pathOrName: string): string | undefined {
432
+ const withoutQuery = pathOrName.split(/[?#]/)[0] ?? ''
433
+ const match = /\.([a-z0-9]+)$/i.exec(withoutQuery)
434
+ return match?.[1]?.toLowerCase()
435
+ }
436
+
437
+ function formatFromMime(mime: string | undefined): string | undefined {
438
+ if (!mime || !mime.startsWith('audio/')) return undefined
439
+ const subtype = mime.slice('audio/'.length).toLowerCase()
440
+ if (subtype === 'mpeg') return 'mp3'
441
+ if (subtype === 'x-wav' || subtype === 'wave') return 'wav'
442
+ return subtype.replace(/^x-/, '')
443
+ }
444
+
445
+ /**
446
+ * Creates a BytePlus Seed Speech transcription adapter with an explicit API
447
+ * key.
448
+ *
449
+ * The key is the **Seed Speech** key, not the Ark key used by the chat, image
450
+ * and video adapters.
451
+ */
452
+ export function createBytePlusTranscription<
453
+ TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,
454
+ >(
455
+ model: TModel,
456
+ apiKey: string,
457
+ config?: Omit<BytePlusVoiceConfig, 'apiKey'>,
458
+ ): BytePlusTranscriptionAdapter<TModel> {
459
+ return new BytePlusTranscriptionAdapter(model, { ...config, apiKey })
460
+ }
461
+
462
+ /**
463
+ * Creates a BytePlus Seed Speech transcription adapter, reading the API key
464
+ * from `BYTEPLUS_VOICE_API_KEY`.
465
+ *
466
+ * @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.
467
+ */
468
+ export function byteplusTranscription<
469
+ TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,
470
+ >(
471
+ model: TModel,
472
+ config?: Omit<BytePlusVoiceConfig, 'apiKey'>,
473
+ ): BytePlusTranscriptionAdapter<TModel> {
474
+ return createBytePlusTranscription(
475
+ model,
476
+ getBytePlusVoiceApiKeyFromEnv(),
477
+ config,
478
+ )
479
+ }