@tanstack/ai-byteplus 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +202 -0
- package/dist/esm/adapters/image.d.ts +89 -0
- package/dist/esm/adapters/image.js +229 -0
- package/dist/esm/adapters/image.js.map +1 -0
- package/dist/esm/adapters/text.d.ts +163 -0
- package/dist/esm/adapters/text.js +347 -0
- package/dist/esm/adapters/text.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +102 -0
- package/dist/esm/adapters/transcription.js +274 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +143 -0
- package/dist/esm/adapters/tts.js +307 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/adapters/video.d.ts +182 -0
- package/dist/esm/adapters/video.js +442 -0
- package/dist/esm/adapters/video.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
- package/dist/esm/audio/tts-provider-options.d.ts +114 -0
- package/dist/esm/audio/wire-types.d.ts +261 -0
- package/dist/esm/audio/wire-types.js +28 -0
- package/dist/esm/audio/wire-types.js.map +1 -0
- package/dist/esm/image/image-provider-options.d.ts +165 -0
- package/dist/esm/image/image-provider-options.js +134 -0
- package/dist/esm/image/image-provider-options.js.map +1 -0
- package/dist/esm/image/wire-types.d.ts +149 -0
- package/dist/esm/index.d.ts +25 -0
- package/dist/esm/index.js +11 -0
- package/dist/esm/message-types.d.ts +154 -0
- package/dist/esm/model-meta.d.ts +594 -0
- package/dist/esm/model-meta.js +619 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/text/text-provider-options.d.ts +109 -0
- package/dist/esm/utils/client.d.ts +183 -0
- package/dist/esm/utils/client.js +253 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +197 -0
- package/dist/esm/video/video-provider-options.js +191 -0
- package/dist/esm/video/video-provider-options.js.map +1 -0
- package/dist/esm/video/wire-types.d.ts +248 -0
- package/package.json +77 -0
- package/src/adapters/image.ts +409 -0
- package/src/adapters/text.ts +539 -0
- package/src/adapters/transcription.ts +479 -0
- package/src/adapters/tts.ts +447 -0
- package/src/adapters/video.ts +732 -0
- package/src/audio/transcription-provider-options.ts +46 -0
- package/src/audio/tts-provider-options.ts +122 -0
- package/src/audio/wire-types.ts +290 -0
- package/src/image/image-provider-options.ts +288 -0
- package/src/image/wire-types.ts +169 -0
- package/src/index.ts +222 -0
- package/src/message-types.ts +169 -0
- package/src/model-meta.ts +954 -0
- package/src/text/text-provider-options.ts +151 -0
- package/src/utils/client.ts +377 -0
- package/src/video/video-provider-options.ts +361 -0
- package/src/video/wire-types.ts +293 -0
|
@@ -0,0 +1,479 @@
|
|
|
1
|
+
import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
|
|
2
|
+
import { arrayBufferToBase64, generateId } from '@tanstack/ai-utils'
|
|
3
|
+
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
4
|
+
import {
|
|
5
|
+
BYTEPLUS_VOICE_BASE_URL,
|
|
6
|
+
bytePlusVoiceError,
|
|
7
|
+
bytePlusVoiceHeaders,
|
|
8
|
+
getBytePlusVoiceApiKeyFromEnv,
|
|
9
|
+
readJsonBody,
|
|
10
|
+
withBytePlusVoiceDefaults,
|
|
11
|
+
} from '../utils/client'
|
|
12
|
+
import {
|
|
13
|
+
BYTEPLUS_ASR_RESOURCE_HEADER,
|
|
14
|
+
BYTEPLUS_ASR_RESOURCE_ID,
|
|
15
|
+
} from '../audio/wire-types'
|
|
16
|
+
import type {
|
|
17
|
+
TokenUsage,
|
|
18
|
+
TranscriptionOptions,
|
|
19
|
+
TranscriptionResult,
|
|
20
|
+
TranscriptionSegment,
|
|
21
|
+
TranscriptionWord,
|
|
22
|
+
} from '@tanstack/ai'
|
|
23
|
+
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
24
|
+
import type { BytePlusVoiceConfig } from '../utils/client'
|
|
25
|
+
import type { BytePlusTranscriptionModel } from '../model-meta'
|
|
26
|
+
import type {
|
|
27
|
+
BytePlusASRAudio,
|
|
28
|
+
BytePlusASRRecognizeRequest,
|
|
29
|
+
BytePlusASRRecognizeResponse,
|
|
30
|
+
BytePlusASRUtterance,
|
|
31
|
+
} from '../audio/wire-types'
|
|
32
|
+
import type { BytePlusTranscriptionProviderOptions } from '../audio/transcription-provider-options'
|
|
33
|
+
|
|
34
|
+
/** Path of the synchronous ("flash") Seed ASR endpoint. */
|
|
35
|
+
const RECOGNIZE_FLASH_PATH = '/api/v3/auc/bigmodel/recognize/flash'
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* BytePlus-specific extension of `TranscriptionWord` carrying the per-word
|
|
39
|
+
* confidence Seed ASR returns. The cross-provider contract has no field for
|
|
40
|
+
* it, so callers who want it narrow the array — the same pattern the Grok
|
|
41
|
+
* adapter uses:
|
|
42
|
+
*
|
|
43
|
+
* ```ts
|
|
44
|
+
* const words = result.words as Array<BytePlusTranscriptionWord> | undefined
|
|
45
|
+
* ```
|
|
46
|
+
*/
|
|
47
|
+
export interface BytePlusTranscriptionWord extends TranscriptionWord {
|
|
48
|
+
/** Model confidence for the word, when Seed ASR returns one. */
|
|
49
|
+
confidence?: number
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Default `user.uid` echoed into BytePlus' request logs. */
|
|
53
|
+
const DEFAULT_UID = 'tanstack-ai'
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* BytePlus Seed Speech transcription (ASR) adapter.
|
|
57
|
+
*
|
|
58
|
+
* Talks to `POST {baseURL}/api/v3/auc/bigmodel/recognize/flash` — the
|
|
59
|
+
* synchronous "flash" endpoint, which returns the whole transcript in one
|
|
60
|
+
* response rather than requiring a submit/poll cycle. It accepts audio up to
|
|
61
|
+
* 2 hours long or 100 MB, either as a publicly reachable URL or as base64
|
|
62
|
+
* bytes.
|
|
63
|
+
*
|
|
64
|
+
* Two BytePlus-specific details:
|
|
65
|
+
*
|
|
66
|
+
* - The model is selected by the `X-Api-Resource-Id` header
|
|
67
|
+
* (`volc.seedasr.auc_turbo`), not by a `model` field in the body. The
|
|
68
|
+
* package's `seed-asr` model id exists to satisfy the SDK contract and to
|
|
69
|
+
* give logs a stable value.
|
|
70
|
+
* - Authentication uses `X-Api-Key` with the **Seed Speech** key, which is a
|
|
71
|
+
* different key from `ARK_API_KEY`.
|
|
72
|
+
*
|
|
73
|
+
* All timings on the wire are milliseconds; they are converted to seconds to
|
|
74
|
+
* match the cross-provider `TranscriptionResult`.
|
|
75
|
+
*
|
|
76
|
+
* @example
|
|
77
|
+
* ```ts
|
|
78
|
+
* const adapter = byteplusTranscription('seed-asr')
|
|
79
|
+
* const result = await generateTranscription({
|
|
80
|
+
* adapter,
|
|
81
|
+
* audio: 'https://example.com/interview.mp3',
|
|
82
|
+
* language: 'en-US',
|
|
83
|
+
* })
|
|
84
|
+
* ```
|
|
85
|
+
*/
|
|
86
|
+
export class BytePlusTranscriptionAdapter<
|
|
87
|
+
TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,
|
|
88
|
+
> extends BaseTranscriptionAdapter<
|
|
89
|
+
TModel,
|
|
90
|
+
BytePlusTranscriptionProviderOptions
|
|
91
|
+
> {
|
|
92
|
+
readonly name = 'byteplus' as const
|
|
93
|
+
|
|
94
|
+
private readonly apiKey: string
|
|
95
|
+
private readonly baseURL: string
|
|
96
|
+
private readonly defaultHeaders: Record<string, string>
|
|
97
|
+
private readonly fetchImpl: typeof fetch
|
|
98
|
+
|
|
99
|
+
constructor(model: TModel, config: BytePlusVoiceConfig) {
|
|
100
|
+
super(model, config)
|
|
101
|
+
const resolved = withBytePlusVoiceDefaults(config)
|
|
102
|
+
this.apiKey = resolved.apiKey
|
|
103
|
+
this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL
|
|
104
|
+
this.defaultHeaders = resolved.defaultHeaders ?? {}
|
|
105
|
+
this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
async transcribe(
|
|
109
|
+
options: TranscriptionOptions<BytePlusTranscriptionProviderOptions>,
|
|
110
|
+
): Promise<TranscriptionResult> {
|
|
111
|
+
const {
|
|
112
|
+
logger,
|
|
113
|
+
model,
|
|
114
|
+
audio,
|
|
115
|
+
language,
|
|
116
|
+
prompt,
|
|
117
|
+
responseFormat,
|
|
118
|
+
modelOptions,
|
|
119
|
+
} = options
|
|
120
|
+
|
|
121
|
+
logger.request(
|
|
122
|
+
`activity=generateTranscription provider=byteplus model=${model}`,
|
|
123
|
+
{ provider: 'byteplus', model },
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
if (prompt) {
|
|
127
|
+
logger.warn(
|
|
128
|
+
'BytePlus Seed ASR has no prompt-biasing field on the flash endpoint — the `prompt` option is ignored.',
|
|
129
|
+
{ provider: 'byteplus', model },
|
|
130
|
+
)
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
// The flash endpoint answers with one JSON shape and offers no format
|
|
134
|
+
// negotiation, so srt/vtt/text/verbose_json can't be honoured. `segments`
|
|
135
|
+
// on the result carry the timings a caller would have wanted from srt/vtt.
|
|
136
|
+
if (responseFormat !== undefined && responseFormat !== 'json') {
|
|
137
|
+
logger.warn(
|
|
138
|
+
`BytePlus Seed ASR always returns JSON — the requested responseFormat "${responseFormat}" is ignored. Build srt/vtt from result.segments if you need them.`,
|
|
139
|
+
{ provider: 'byteplus', model, responseFormat },
|
|
140
|
+
)
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
try {
|
|
144
|
+
const audioPayload = await normalizeAudioInput(
|
|
145
|
+
audio,
|
|
146
|
+
modelOptions?.audio_format,
|
|
147
|
+
)
|
|
148
|
+
const body = buildRecognizeRequestBody({
|
|
149
|
+
audio: audioPayload,
|
|
150
|
+
language,
|
|
151
|
+
modelOptions,
|
|
152
|
+
})
|
|
153
|
+
|
|
154
|
+
const response = await this.fetchImpl(
|
|
155
|
+
`${this.baseURL}${RECOGNIZE_FLASH_PATH}`,
|
|
156
|
+
{
|
|
157
|
+
method: 'POST',
|
|
158
|
+
headers: bytePlusVoiceHeaders(this.apiKey, {
|
|
159
|
+
...this.defaultHeaders,
|
|
160
|
+
[BYTEPLUS_ASR_RESOURCE_HEADER]: BYTEPLUS_ASR_RESOURCE_ID,
|
|
161
|
+
}),
|
|
162
|
+
body: JSON.stringify(body),
|
|
163
|
+
},
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
const payload = await readJsonBody(response)
|
|
167
|
+
|
|
168
|
+
if (!response.ok) {
|
|
169
|
+
throw bytePlusVoiceError(response.status, payload, 'transcription')
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
const data = payload as BytePlusASRRecognizeResponse
|
|
173
|
+
const text = data.result?.text ?? data.transcript
|
|
174
|
+
|
|
175
|
+
// The flash endpoint can answer HTTP 200 while carrying the numeric
|
|
176
|
+
// error envelope, so an absent transcript is a failure rather than an
|
|
177
|
+
// empty result.
|
|
178
|
+
if (typeof text !== 'string') {
|
|
179
|
+
throw bytePlusVoiceError(response.status, payload, 'transcription')
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
// An empty string is well-formed, so it isn't an error — silence is a
|
|
183
|
+
// legitimate transcription. But it is also what a 200-wrapped failure
|
|
184
|
+
// looks like, so say so rather than handing back a successful, empty
|
|
185
|
+
// result with no signal.
|
|
186
|
+
if (text === '' && !hasUtterances(data)) {
|
|
187
|
+
logger.warn(
|
|
188
|
+
`byteplus: transcription returned an empty transcript with no ` +
|
|
189
|
+
`utterances. This is a valid result for silent audio, and is also ` +
|
|
190
|
+
`what a 200-wrapped failure looks like.`,
|
|
191
|
+
{ provider: this.name, model },
|
|
192
|
+
)
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// Seed ASR doesn't echo the language back, so report the one that was
|
|
196
|
+
// actually sent — which is `modelOptions.language` when it overrode the
|
|
197
|
+
// cross-provider hint.
|
|
198
|
+
const requestedLanguage = modelOptions?.language ?? language
|
|
199
|
+
|
|
200
|
+
return {
|
|
201
|
+
id: generateId(this.name),
|
|
202
|
+
model,
|
|
203
|
+
...mapRecognizeResponse(data, text, logger),
|
|
204
|
+
...(requestedLanguage !== undefined && { language: requestedLanguage }),
|
|
205
|
+
}
|
|
206
|
+
} catch (error) {
|
|
207
|
+
logger.errors('byteplus.transcribe fatal', {
|
|
208
|
+
error: toRunErrorPayload(error, 'byteplus.transcribe failed'),
|
|
209
|
+
source: 'byteplus.transcribe',
|
|
210
|
+
})
|
|
211
|
+
throw error
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
/**
|
|
217
|
+
* Build the JSON body for `POST /api/v3/auc/bigmodel/recognize/flash`.
|
|
218
|
+
*
|
|
219
|
+
* `show_utterances` defaults to `true` so the response carries the
|
|
220
|
+
* per-utterance breakdown that populates `segments` and `words`.
|
|
221
|
+
*/
|
|
222
|
+
export function buildRecognizeRequestBody(options: {
|
|
223
|
+
audio: BytePlusASRAudio
|
|
224
|
+
language: string | undefined
|
|
225
|
+
modelOptions: BytePlusTranscriptionProviderOptions | undefined
|
|
226
|
+
}): BytePlusASRRecognizeRequest {
|
|
227
|
+
const { audio, language, modelOptions } = options
|
|
228
|
+
|
|
229
|
+
const resolvedLanguage = modelOptions?.language ?? language
|
|
230
|
+
|
|
231
|
+
return {
|
|
232
|
+
user: { uid: modelOptions?.uid ?? DEFAULT_UID },
|
|
233
|
+
audio,
|
|
234
|
+
request: {
|
|
235
|
+
model_name: modelOptions?.model_name ?? 'bigmodel',
|
|
236
|
+
show_utterances: modelOptions?.show_utterances ?? true,
|
|
237
|
+
...(modelOptions?.enable_itn !== undefined && {
|
|
238
|
+
enable_itn: modelOptions.enable_itn,
|
|
239
|
+
}),
|
|
240
|
+
...(modelOptions?.enable_punc !== undefined && {
|
|
241
|
+
enable_punc: modelOptions.enable_punc,
|
|
242
|
+
}),
|
|
243
|
+
...(modelOptions?.enable_ddc !== undefined && {
|
|
244
|
+
enable_ddc: modelOptions.enable_ddc,
|
|
245
|
+
}),
|
|
246
|
+
...(modelOptions?.enable_speaker_info !== undefined && {
|
|
247
|
+
enable_speaker_info: modelOptions.enable_speaker_info,
|
|
248
|
+
}),
|
|
249
|
+
...(resolvedLanguage !== undefined && { language: resolvedLanguage }),
|
|
250
|
+
},
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* Turn a recognition response into the transcript-shaped half of a
|
|
256
|
+
* `TranscriptionResult`. Wire timings are milliseconds; everything returned
|
|
257
|
+
* here is seconds.
|
|
258
|
+
*/
|
|
259
|
+
export function mapRecognizeResponse(
|
|
260
|
+
data: BytePlusASRRecognizeResponse,
|
|
261
|
+
text: string,
|
|
262
|
+
logger?: InternalLogger,
|
|
263
|
+
): Omit<TranscriptionResult, 'id' | 'model'> {
|
|
264
|
+
const utterances = data.result?.utterances ?? data.utterances ?? []
|
|
265
|
+
// `id` numbers the segments we emit, not the utterances we were given, so
|
|
266
|
+
// dropping an untimed utterance doesn't leave a hole in the sequence.
|
|
267
|
+
const segments = utterances
|
|
268
|
+
.flatMap((utterance) => toSegment(utterance))
|
|
269
|
+
.map((segment, index) => ({ ...segment, id: index }))
|
|
270
|
+
|
|
271
|
+
const rawWords = utterances.flatMap((utterance) => utterance.words ?? [])
|
|
272
|
+
const words = rawWords.flatMap((word) => {
|
|
273
|
+
if (
|
|
274
|
+
typeof word.text !== 'string' ||
|
|
275
|
+
typeof word.start_time !== 'number' ||
|
|
276
|
+
typeof word.end_time !== 'number'
|
|
277
|
+
) {
|
|
278
|
+
return []
|
|
279
|
+
}
|
|
280
|
+
const mapped: BytePlusTranscriptionWord = {
|
|
281
|
+
word: word.text,
|
|
282
|
+
start: msToSeconds(word.start_time),
|
|
283
|
+
end: msToSeconds(word.end_time),
|
|
284
|
+
}
|
|
285
|
+
if (word.confidence !== undefined) mapped.confidence = word.confidence
|
|
286
|
+
return [mapped]
|
|
287
|
+
})
|
|
288
|
+
|
|
289
|
+
// Untimed entries are dropped rather than emitted with NaN timings, but a
|
|
290
|
+
// silent drop leaves the caller unable to tell "the provider sent no
|
|
291
|
+
// timings" from "the adapter discarded them" — the two have very different
|
|
292
|
+
// fixes, and a field rename upstream (e.g. `text` → `word`) would empty
|
|
293
|
+
// these arrays without a single error anywhere.
|
|
294
|
+
const droppedWords = rawWords.length - words.length
|
|
295
|
+
if (droppedWords > 0) {
|
|
296
|
+
logger?.warn(
|
|
297
|
+
`byteplus: dropped ${droppedWords} of ${rawWords.length} word(s) with ` +
|
|
298
|
+
`missing or non-numeric timings.`,
|
|
299
|
+
{ provider: 'byteplus' },
|
|
300
|
+
)
|
|
301
|
+
}
|
|
302
|
+
const droppedSegments = utterances.length - segments.length
|
|
303
|
+
if (droppedSegments > 0) {
|
|
304
|
+
logger?.warn(
|
|
305
|
+
`byteplus: dropped ${droppedSegments} of ${utterances.length} ` +
|
|
306
|
+
`utterance(s) with missing or non-numeric timings.`,
|
|
307
|
+
{ provider: 'byteplus' },
|
|
308
|
+
)
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
const durationMs = data.audio_info?.duration
|
|
312
|
+
const duration =
|
|
313
|
+
typeof durationMs === 'number' && durationMs > 0
|
|
314
|
+
? msToSeconds(durationMs)
|
|
315
|
+
: undefined
|
|
316
|
+
|
|
317
|
+
// Seed ASR is duration-billed and reports no token counts, so `usage`
|
|
318
|
+
// carries only the audio length — the same shape the Grok and OpenAI
|
|
319
|
+
// whisper paths use.
|
|
320
|
+
const usage: TokenUsage | undefined =
|
|
321
|
+
duration !== undefined
|
|
322
|
+
? {
|
|
323
|
+
promptTokens: 0,
|
|
324
|
+
completionTokens: 0,
|
|
325
|
+
totalTokens: 0,
|
|
326
|
+
durationSeconds: duration,
|
|
327
|
+
}
|
|
328
|
+
: undefined
|
|
329
|
+
|
|
330
|
+
return {
|
|
331
|
+
text,
|
|
332
|
+
...(duration !== undefined && { duration }),
|
|
333
|
+
...(segments.length > 0 && { segments }),
|
|
334
|
+
...(words.length > 0 && { words }),
|
|
335
|
+
...(usage !== undefined && { usage }),
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
/**
|
|
340
|
+
* Convert one utterance into a segment, or nothing when it carries no
|
|
341
|
+
* timings. The `id` is a placeholder — the caller renumbers after filtering.
|
|
342
|
+
*/
|
|
343
|
+
/**
|
|
344
|
+
* True when the response carries at least one utterance, in either envelope
|
|
345
|
+
* form. Used to tell "silent audio" from a 200-wrapped failure: a genuinely
|
|
346
|
+
* empty transcript usually still arrives with no utterances, so the pairing is
|
|
347
|
+
* a hint rather than proof — hence a warning rather than a throw.
|
|
348
|
+
*/
|
|
349
|
+
function hasUtterances(data: BytePlusASRRecognizeResponse): boolean {
|
|
350
|
+
return (data.result?.utterances ?? data.utterances ?? []).length > 0
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
function toSegment(
|
|
354
|
+
utterance: BytePlusASRUtterance,
|
|
355
|
+
): Array<TranscriptionSegment> {
|
|
356
|
+
if (
|
|
357
|
+
typeof utterance.start_time !== 'number' ||
|
|
358
|
+
typeof utterance.end_time !== 'number'
|
|
359
|
+
) {
|
|
360
|
+
return []
|
|
361
|
+
}
|
|
362
|
+
const speaker = utterance.additions?.speaker
|
|
363
|
+
return [
|
|
364
|
+
{
|
|
365
|
+
id: 0,
|
|
366
|
+
start: msToSeconds(utterance.start_time),
|
|
367
|
+
end: msToSeconds(utterance.end_time),
|
|
368
|
+
text: utterance.text ?? '',
|
|
369
|
+
...(speaker !== undefined && { speaker }),
|
|
370
|
+
},
|
|
371
|
+
]
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* **Must verify when the Seed Speech key lands.** Every timing this adapter
|
|
376
|
+
* reads — `audio_info.duration`, and each utterance's and word's
|
|
377
|
+
* `start_time` / `end_time` — is assumed to be milliseconds. That comes from
|
|
378
|
+
* the Volcengine flash-recognition reference this endpoint derives from
|
|
379
|
+
* (a 2.499 s clip reports `duration: 2499`), not from a BytePlus response we
|
|
380
|
+
* have seen. If BytePlus reports seconds instead, every duration, segment and
|
|
381
|
+
* word timing here is 1000× too small, and this is the only place to fix.
|
|
382
|
+
*/
|
|
383
|
+
function msToSeconds(milliseconds: number): number {
|
|
384
|
+
return milliseconds / 1000
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
/**
|
|
388
|
+
* Turn the cross-provider `audio` input into the endpoint's `audio` block.
|
|
389
|
+
*
|
|
390
|
+
* URLs are passed through untouched — Seed ASR fetches them itself, which
|
|
391
|
+
* avoids pulling large media through this process. Everything else is sent as
|
|
392
|
+
* base64 `data`, with the container inferred from the input's MIME type or
|
|
393
|
+
* filename when the caller didn't pin `audio_format`.
|
|
394
|
+
*/
|
|
395
|
+
export async function normalizeAudioInput(
|
|
396
|
+
audio: TranscriptionOptions['audio'],
|
|
397
|
+
formatHint: string | undefined,
|
|
398
|
+
): Promise<BytePlusASRAudio> {
|
|
399
|
+
const withFormat = (
|
|
400
|
+
payload: BytePlusASRAudio,
|
|
401
|
+
inferred?: string,
|
|
402
|
+
): BytePlusASRAudio => {
|
|
403
|
+
const format = formatHint ?? inferred
|
|
404
|
+
return format ? { ...payload, format } : payload
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
if (typeof audio === 'string') {
|
|
408
|
+
if (/^https?:\/\//i.test(audio)) {
|
|
409
|
+
return withFormat({ url: audio }, extensionOf(audio))
|
|
410
|
+
}
|
|
411
|
+
const dataUrl = /^data:([^;,]+)?(?:;[^,]*)*,(.*)$/s.exec(audio)
|
|
412
|
+
if (dataUrl) {
|
|
413
|
+
return withFormat({ data: dataUrl[2] ?? '' }, formatFromMime(dataUrl[1]))
|
|
414
|
+
}
|
|
415
|
+
// A bare string that is neither a URL nor a data URL is already base64.
|
|
416
|
+
return withFormat({ data: audio })
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
if (audio instanceof ArrayBuffer) {
|
|
420
|
+
return withFormat({ data: arrayBufferToBase64(audio) })
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
const data = arrayBufferToBase64(await audio.arrayBuffer())
|
|
424
|
+
const inferred =
|
|
425
|
+
('name' in audio && typeof audio.name === 'string'
|
|
426
|
+
? extensionOf(audio.name)
|
|
427
|
+
: undefined) ?? formatFromMime(audio.type)
|
|
428
|
+
return withFormat({ data }, inferred)
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
function extensionOf(pathOrName: string): string | undefined {
|
|
432
|
+
const withoutQuery = pathOrName.split(/[?#]/)[0] ?? ''
|
|
433
|
+
const match = /\.([a-z0-9]+)$/i.exec(withoutQuery)
|
|
434
|
+
return match?.[1]?.toLowerCase()
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
function formatFromMime(mime: string | undefined): string | undefined {
|
|
438
|
+
if (!mime || !mime.startsWith('audio/')) return undefined
|
|
439
|
+
const subtype = mime.slice('audio/'.length).toLowerCase()
|
|
440
|
+
if (subtype === 'mpeg') return 'mp3'
|
|
441
|
+
if (subtype === 'x-wav' || subtype === 'wave') return 'wav'
|
|
442
|
+
return subtype.replace(/^x-/, '')
|
|
443
|
+
}
|
|
444
|
+
|
|
445
|
+
/**
|
|
446
|
+
* Creates a BytePlus Seed Speech transcription adapter with an explicit API
|
|
447
|
+
* key.
|
|
448
|
+
*
|
|
449
|
+
* The key is the **Seed Speech** key, not the Ark key used by the chat, image
|
|
450
|
+
* and video adapters.
|
|
451
|
+
*/
|
|
452
|
+
export function createBytePlusTranscription<
|
|
453
|
+
TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,
|
|
454
|
+
>(
|
|
455
|
+
model: TModel,
|
|
456
|
+
apiKey: string,
|
|
457
|
+
config?: Omit<BytePlusVoiceConfig, 'apiKey'>,
|
|
458
|
+
): BytePlusTranscriptionAdapter<TModel> {
|
|
459
|
+
return new BytePlusTranscriptionAdapter(model, { ...config, apiKey })
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
/**
|
|
463
|
+
* Creates a BytePlus Seed Speech transcription adapter, reading the API key
|
|
464
|
+
* from `BYTEPLUS_VOICE_API_KEY`.
|
|
465
|
+
*
|
|
466
|
+
* @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.
|
|
467
|
+
*/
|
|
468
|
+
export function byteplusTranscription<
|
|
469
|
+
TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,
|
|
470
|
+
>(
|
|
471
|
+
model: TModel,
|
|
472
|
+
config?: Omit<BytePlusVoiceConfig, 'apiKey'>,
|
|
473
|
+
): BytePlusTranscriptionAdapter<TModel> {
|
|
474
|
+
return createBytePlusTranscription(
|
|
475
|
+
model,
|
|
476
|
+
getBytePlusVoiceApiKeyFromEnv(),
|
|
477
|
+
config,
|
|
478
|
+
)
|
|
479
|
+
}
|