@tanstack/ai-byteplus 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +202 -0
- package/dist/esm/adapters/image.d.ts +89 -0
- package/dist/esm/adapters/image.js +229 -0
- package/dist/esm/adapters/image.js.map +1 -0
- package/dist/esm/adapters/text.d.ts +163 -0
- package/dist/esm/adapters/text.js +347 -0
- package/dist/esm/adapters/text.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +102 -0
- package/dist/esm/adapters/transcription.js +274 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +143 -0
- package/dist/esm/adapters/tts.js +307 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/adapters/video.d.ts +182 -0
- package/dist/esm/adapters/video.js +442 -0
- package/dist/esm/adapters/video.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
- package/dist/esm/audio/tts-provider-options.d.ts +114 -0
- package/dist/esm/audio/wire-types.d.ts +261 -0
- package/dist/esm/audio/wire-types.js +28 -0
- package/dist/esm/audio/wire-types.js.map +1 -0
- package/dist/esm/image/image-provider-options.d.ts +165 -0
- package/dist/esm/image/image-provider-options.js +134 -0
- package/dist/esm/image/image-provider-options.js.map +1 -0
- package/dist/esm/image/wire-types.d.ts +149 -0
- package/dist/esm/index.d.ts +25 -0
- package/dist/esm/index.js +11 -0
- package/dist/esm/message-types.d.ts +154 -0
- package/dist/esm/model-meta.d.ts +594 -0
- package/dist/esm/model-meta.js +619 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/text/text-provider-options.d.ts +109 -0
- package/dist/esm/utils/client.d.ts +183 -0
- package/dist/esm/utils/client.js +253 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +197 -0
- package/dist/esm/video/video-provider-options.js +191 -0
- package/dist/esm/video/video-provider-options.js.map +1 -0
- package/dist/esm/video/wire-types.d.ts +248 -0
- package/package.json +77 -0
- package/src/adapters/image.ts +409 -0
- package/src/adapters/text.ts +539 -0
- package/src/adapters/transcription.ts +479 -0
- package/src/adapters/tts.ts +447 -0
- package/src/adapters/video.ts +732 -0
- package/src/audio/transcription-provider-options.ts +46 -0
- package/src/audio/tts-provider-options.ts +122 -0
- package/src/audio/wire-types.ts +290 -0
- package/src/image/image-provider-options.ts +288 -0
- package/src/image/wire-types.ts +169 -0
- package/src/index.ts +222 -0
- package/src/message-types.ts +169 -0
- package/src/model-meta.ts +954 -0
- package/src/text/text-provider-options.ts +151 -0
- package/src/utils/client.ts +377 -0
- package/src/video/video-provider-options.ts +361 -0
- package/src/video/wire-types.ts +293 -0
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
import { bytePlusVoiceError, bytePlusVoiceHeaders, getBytePlusVoiceApiKeyFromEnv, readJsonBody, withBytePlusVoiceDefaults } from "../utils/client.js";
|
|
2
|
+
import { BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID } from "../audio/wire-types.js";
|
|
3
|
+
import { BaseTranscriptionAdapter } from "@tanstack/ai/adapters";
|
|
4
|
+
import { toRunErrorPayload } from "@tanstack/ai/adapter-internals";
|
|
5
|
+
import { arrayBufferToBase64, generateId } from "@tanstack/ai-utils";
|
|
6
|
+
//#region src/adapters/transcription.ts
|
|
7
|
+
/** Path of the synchronous ("flash") Seed ASR endpoint. */
|
|
8
|
+
var RECOGNIZE_FLASH_PATH = "/api/v3/auc/bigmodel/recognize/flash";
|
|
9
|
+
/** Default `user.uid` echoed into BytePlus' request logs. */
|
|
10
|
+
var DEFAULT_UID = "tanstack-ai";
|
|
11
|
+
/**
|
|
12
|
+
* BytePlus Seed Speech transcription (ASR) adapter.
|
|
13
|
+
*
|
|
14
|
+
* Talks to `POST {baseURL}/api/v3/auc/bigmodel/recognize/flash` — the
|
|
15
|
+
* synchronous "flash" endpoint, which returns the whole transcript in one
|
|
16
|
+
* response rather than requiring a submit/poll cycle. It accepts audio up to
|
|
17
|
+
* 2 hours long or 100 MB, either as a publicly reachable URL or as base64
|
|
18
|
+
* bytes.
|
|
19
|
+
*
|
|
20
|
+
* Two BytePlus-specific details:
|
|
21
|
+
*
|
|
22
|
+
* - The model is selected by the `X-Api-Resource-Id` header
|
|
23
|
+
* (`volc.seedasr.auc_turbo`), not by a `model` field in the body. The
|
|
24
|
+
* package's `seed-asr` model id exists to satisfy the SDK contract and to
|
|
25
|
+
* give logs a stable value.
|
|
26
|
+
* - Authentication uses `X-Api-Key` with the **Seed Speech** key, which is a
|
|
27
|
+
* different key from `ARK_API_KEY`.
|
|
28
|
+
*
|
|
29
|
+
* All timings on the wire are milliseconds; they are converted to seconds to
|
|
30
|
+
* match the cross-provider `TranscriptionResult`.
|
|
31
|
+
*
|
|
32
|
+
* @example
|
|
33
|
+
* ```ts
|
|
34
|
+
* const adapter = byteplusTranscription('seed-asr')
|
|
35
|
+
* const result = await generateTranscription({
|
|
36
|
+
* adapter,
|
|
37
|
+
* audio: 'https://example.com/interview.mp3',
|
|
38
|
+
* language: 'en-US',
|
|
39
|
+
* })
|
|
40
|
+
* ```
|
|
41
|
+
*/
|
|
42
|
+
var BytePlusTranscriptionAdapter = class extends BaseTranscriptionAdapter {
|
|
43
|
+
name = "byteplus";
|
|
44
|
+
apiKey;
|
|
45
|
+
baseURL;
|
|
46
|
+
defaultHeaders;
|
|
47
|
+
fetchImpl;
|
|
48
|
+
constructor(model, config) {
|
|
49
|
+
super(model, config);
|
|
50
|
+
const resolved = withBytePlusVoiceDefaults(config);
|
|
51
|
+
this.apiKey = resolved.apiKey;
|
|
52
|
+
this.baseURL = resolved.baseURL ?? "https://voice.ap-southeast-1.bytepluses.com";
|
|
53
|
+
this.defaultHeaders = resolved.defaultHeaders ?? {};
|
|
54
|
+
this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis);
|
|
55
|
+
}
|
|
56
|
+
async transcribe(options) {
|
|
57
|
+
const { logger, model, audio, language, prompt, responseFormat, modelOptions } = options;
|
|
58
|
+
logger.request(`activity=generateTranscription provider=byteplus model=${model}`, {
|
|
59
|
+
provider: "byteplus",
|
|
60
|
+
model
|
|
61
|
+
});
|
|
62
|
+
if (prompt) logger.warn("BytePlus Seed ASR has no prompt-biasing field on the flash endpoint — the `prompt` option is ignored.", {
|
|
63
|
+
provider: "byteplus",
|
|
64
|
+
model
|
|
65
|
+
});
|
|
66
|
+
if (responseFormat !== void 0 && responseFormat !== "json") logger.warn(`BytePlus Seed ASR always returns JSON — the requested responseFormat "${responseFormat}" is ignored. Build srt/vtt from result.segments if you need them.`, {
|
|
67
|
+
provider: "byteplus",
|
|
68
|
+
model,
|
|
69
|
+
responseFormat
|
|
70
|
+
});
|
|
71
|
+
try {
|
|
72
|
+
const body = buildRecognizeRequestBody({
|
|
73
|
+
audio: await normalizeAudioInput(audio, modelOptions?.audio_format),
|
|
74
|
+
language,
|
|
75
|
+
modelOptions
|
|
76
|
+
});
|
|
77
|
+
const response = await this.fetchImpl(`${this.baseURL}${RECOGNIZE_FLASH_PATH}`, {
|
|
78
|
+
method: "POST",
|
|
79
|
+
headers: bytePlusVoiceHeaders(this.apiKey, {
|
|
80
|
+
...this.defaultHeaders,
|
|
81
|
+
[BYTEPLUS_ASR_RESOURCE_HEADER]: BYTEPLUS_ASR_RESOURCE_ID
|
|
82
|
+
}),
|
|
83
|
+
body: JSON.stringify(body)
|
|
84
|
+
});
|
|
85
|
+
const payload = await readJsonBody(response);
|
|
86
|
+
if (!response.ok) throw bytePlusVoiceError(response.status, payload, "transcription");
|
|
87
|
+
const data = payload;
|
|
88
|
+
const text = data.result?.text ?? data.transcript;
|
|
89
|
+
if (typeof text !== "string") throw bytePlusVoiceError(response.status, payload, "transcription");
|
|
90
|
+
if (text === "" && !hasUtterances(data)) logger.warn("byteplus: transcription returned an empty transcript with no utterances. This is a valid result for silent audio, and is also what a 200-wrapped failure looks like.", {
|
|
91
|
+
provider: this.name,
|
|
92
|
+
model
|
|
93
|
+
});
|
|
94
|
+
const requestedLanguage = modelOptions?.language ?? language;
|
|
95
|
+
return {
|
|
96
|
+
id: generateId(this.name),
|
|
97
|
+
model,
|
|
98
|
+
...mapRecognizeResponse(data, text, logger),
|
|
99
|
+
...requestedLanguage !== void 0 && { language: requestedLanguage }
|
|
100
|
+
};
|
|
101
|
+
} catch (error) {
|
|
102
|
+
logger.errors("byteplus.transcribe fatal", {
|
|
103
|
+
error: toRunErrorPayload(error, "byteplus.transcribe failed"),
|
|
104
|
+
source: "byteplus.transcribe"
|
|
105
|
+
});
|
|
106
|
+
throw error;
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
};
|
|
110
|
+
/**
|
|
111
|
+
* Build the JSON body for `POST /api/v3/auc/bigmodel/recognize/flash`.
|
|
112
|
+
*
|
|
113
|
+
* `show_utterances` defaults to `true` so the response carries the
|
|
114
|
+
* per-utterance breakdown that populates `segments` and `words`.
|
|
115
|
+
*/
|
|
116
|
+
function buildRecognizeRequestBody(options) {
|
|
117
|
+
const { audio, language, modelOptions } = options;
|
|
118
|
+
const resolvedLanguage = modelOptions?.language ?? language;
|
|
119
|
+
return {
|
|
120
|
+
user: { uid: modelOptions?.uid ?? DEFAULT_UID },
|
|
121
|
+
audio,
|
|
122
|
+
request: {
|
|
123
|
+
model_name: modelOptions?.model_name ?? "bigmodel",
|
|
124
|
+
show_utterances: modelOptions?.show_utterances ?? true,
|
|
125
|
+
...modelOptions?.enable_itn !== void 0 && { enable_itn: modelOptions.enable_itn },
|
|
126
|
+
...modelOptions?.enable_punc !== void 0 && { enable_punc: modelOptions.enable_punc },
|
|
127
|
+
...modelOptions?.enable_ddc !== void 0 && { enable_ddc: modelOptions.enable_ddc },
|
|
128
|
+
...modelOptions?.enable_speaker_info !== void 0 && { enable_speaker_info: modelOptions.enable_speaker_info },
|
|
129
|
+
...resolvedLanguage !== void 0 && { language: resolvedLanguage }
|
|
130
|
+
}
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Turn a recognition response into the transcript-shaped half of a
|
|
135
|
+
* `TranscriptionResult`. Wire timings are milliseconds; everything returned
|
|
136
|
+
* here is seconds.
|
|
137
|
+
*/
|
|
138
|
+
function mapRecognizeResponse(data, text, logger) {
|
|
139
|
+
const utterances = data.result?.utterances ?? data.utterances ?? [];
|
|
140
|
+
const segments = utterances.flatMap((utterance) => toSegment(utterance)).map((segment, index) => ({
|
|
141
|
+
...segment,
|
|
142
|
+
id: index
|
|
143
|
+
}));
|
|
144
|
+
const rawWords = utterances.flatMap((utterance) => utterance.words ?? []);
|
|
145
|
+
const words = rawWords.flatMap((word) => {
|
|
146
|
+
if (typeof word.text !== "string" || typeof word.start_time !== "number" || typeof word.end_time !== "number") return [];
|
|
147
|
+
const mapped = {
|
|
148
|
+
word: word.text,
|
|
149
|
+
start: msToSeconds(word.start_time),
|
|
150
|
+
end: msToSeconds(word.end_time)
|
|
151
|
+
};
|
|
152
|
+
if (word.confidence !== void 0) mapped.confidence = word.confidence;
|
|
153
|
+
return [mapped];
|
|
154
|
+
});
|
|
155
|
+
const droppedWords = rawWords.length - words.length;
|
|
156
|
+
if (droppedWords > 0) logger?.warn(`byteplus: dropped ${droppedWords} of ${rawWords.length} word(s) with missing or non-numeric timings.`, { provider: "byteplus" });
|
|
157
|
+
const droppedSegments = utterances.length - segments.length;
|
|
158
|
+
if (droppedSegments > 0) logger?.warn(`byteplus: dropped ${droppedSegments} of ${utterances.length} utterance(s) with missing or non-numeric timings.`, { provider: "byteplus" });
|
|
159
|
+
const durationMs = data.audio_info?.duration;
|
|
160
|
+
const duration = typeof durationMs === "number" && durationMs > 0 ? msToSeconds(durationMs) : void 0;
|
|
161
|
+
const usage = duration !== void 0 ? {
|
|
162
|
+
promptTokens: 0,
|
|
163
|
+
completionTokens: 0,
|
|
164
|
+
totalTokens: 0,
|
|
165
|
+
durationSeconds: duration
|
|
166
|
+
} : void 0;
|
|
167
|
+
return {
|
|
168
|
+
text,
|
|
169
|
+
...duration !== void 0 && { duration },
|
|
170
|
+
...segments.length > 0 && { segments },
|
|
171
|
+
...words.length > 0 && { words },
|
|
172
|
+
...usage !== void 0 && { usage }
|
|
173
|
+
};
|
|
174
|
+
}
|
|
175
|
+
/**
|
|
176
|
+
* Convert one utterance into a segment, or nothing when it carries no
|
|
177
|
+
* timings. The `id` is a placeholder — the caller renumbers after filtering.
|
|
178
|
+
*/
|
|
179
|
+
/**
|
|
180
|
+
* True when the response carries at least one utterance, in either envelope
|
|
181
|
+
* form. Used to tell "silent audio" from a 200-wrapped failure: a genuinely
|
|
182
|
+
* empty transcript usually still arrives with no utterances, so the pairing is
|
|
183
|
+
* a hint rather than proof — hence a warning rather than a throw.
|
|
184
|
+
*/
|
|
185
|
+
function hasUtterances(data) {
|
|
186
|
+
return (data.result?.utterances ?? data.utterances ?? []).length > 0;
|
|
187
|
+
}
|
|
188
|
+
function toSegment(utterance) {
|
|
189
|
+
if (typeof utterance.start_time !== "number" || typeof utterance.end_time !== "number") return [];
|
|
190
|
+
const speaker = utterance.additions?.speaker;
|
|
191
|
+
return [{
|
|
192
|
+
id: 0,
|
|
193
|
+
start: msToSeconds(utterance.start_time),
|
|
194
|
+
end: msToSeconds(utterance.end_time),
|
|
195
|
+
text: utterance.text ?? "",
|
|
196
|
+
...speaker !== void 0 && { speaker }
|
|
197
|
+
}];
|
|
198
|
+
}
|
|
199
|
+
/**
|
|
200
|
+
* **Must verify when the Seed Speech key lands.** Every timing this adapter
|
|
201
|
+
* reads — `audio_info.duration`, and each utterance's and word's
|
|
202
|
+
* `start_time` / `end_time` — is assumed to be milliseconds. That comes from
|
|
203
|
+
* the Volcengine flash-recognition reference this endpoint derives from
|
|
204
|
+
* (a 2.499 s clip reports `duration: 2499`), not from a BytePlus response we
|
|
205
|
+
* have seen. If BytePlus reports seconds instead, every duration, segment and
|
|
206
|
+
* word timing here is 1000× too small, and this is the only place to fix.
|
|
207
|
+
*/
|
|
208
|
+
function msToSeconds(milliseconds) {
|
|
209
|
+
return milliseconds / 1e3;
|
|
210
|
+
}
|
|
211
|
+
/**
|
|
212
|
+
* Turn the cross-provider `audio` input into the endpoint's `audio` block.
|
|
213
|
+
*
|
|
214
|
+
* URLs are passed through untouched — Seed ASR fetches them itself, which
|
|
215
|
+
* avoids pulling large media through this process. Everything else is sent as
|
|
216
|
+
* base64 `data`, with the container inferred from the input's MIME type or
|
|
217
|
+
* filename when the caller didn't pin `audio_format`.
|
|
218
|
+
*/
|
|
219
|
+
async function normalizeAudioInput(audio, formatHint) {
|
|
220
|
+
const withFormat = (payload, inferred) => {
|
|
221
|
+
const format = formatHint ?? inferred;
|
|
222
|
+
return format ? {
|
|
223
|
+
...payload,
|
|
224
|
+
format
|
|
225
|
+
} : payload;
|
|
226
|
+
};
|
|
227
|
+
if (typeof audio === "string") {
|
|
228
|
+
if (/^https?:\/\//i.test(audio)) return withFormat({ url: audio }, extensionOf(audio));
|
|
229
|
+
const dataUrl = /^data:([^;,]+)?(?:;[^,]*)*,(.*)$/s.exec(audio);
|
|
230
|
+
if (dataUrl) return withFormat({ data: dataUrl[2] ?? "" }, formatFromMime(dataUrl[1]));
|
|
231
|
+
return withFormat({ data: audio });
|
|
232
|
+
}
|
|
233
|
+
if (audio instanceof ArrayBuffer) return withFormat({ data: arrayBufferToBase64(audio) });
|
|
234
|
+
const data = arrayBufferToBase64(await audio.arrayBuffer());
|
|
235
|
+
const inferred = ("name" in audio && typeof audio.name === "string" ? extensionOf(audio.name) : void 0) ?? formatFromMime(audio.type);
|
|
236
|
+
return withFormat({ data }, inferred);
|
|
237
|
+
}
|
|
238
|
+
function extensionOf(pathOrName) {
|
|
239
|
+
const withoutQuery = pathOrName.split(/[?#]/)[0] ?? "";
|
|
240
|
+
return /\.([a-z0-9]+)$/i.exec(withoutQuery)?.[1]?.toLowerCase();
|
|
241
|
+
}
|
|
242
|
+
function formatFromMime(mime) {
|
|
243
|
+
if (!mime || !mime.startsWith("audio/")) return void 0;
|
|
244
|
+
const subtype = mime.slice(6).toLowerCase();
|
|
245
|
+
if (subtype === "mpeg") return "mp3";
|
|
246
|
+
if (subtype === "x-wav" || subtype === "wave") return "wav";
|
|
247
|
+
return subtype.replace(/^x-/, "");
|
|
248
|
+
}
|
|
249
|
+
/**
|
|
250
|
+
* Creates a BytePlus Seed Speech transcription adapter with an explicit API
|
|
251
|
+
* key.
|
|
252
|
+
*
|
|
253
|
+
* The key is the **Seed Speech** key, not the Ark key used by the chat, image
|
|
254
|
+
* and video adapters.
|
|
255
|
+
*/
|
|
256
|
+
function createBytePlusTranscription(model, apiKey, config) {
|
|
257
|
+
return new BytePlusTranscriptionAdapter(model, {
|
|
258
|
+
...config,
|
|
259
|
+
apiKey
|
|
260
|
+
});
|
|
261
|
+
}
|
|
262
|
+
/**
|
|
263
|
+
* Creates a BytePlus Seed Speech transcription adapter, reading the API key
|
|
264
|
+
* from `BYTEPLUS_VOICE_API_KEY`.
|
|
265
|
+
*
|
|
266
|
+
* @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.
|
|
267
|
+
*/
|
|
268
|
+
function byteplusTranscription(model, config) {
|
|
269
|
+
return createBytePlusTranscription(model, getBytePlusVoiceApiKeyFromEnv(), config);
|
|
270
|
+
}
|
|
271
|
+
//#endregion
|
|
272
|
+
export { BytePlusTranscriptionAdapter, buildRecognizeRequestBody, byteplusTranscription, createBytePlusTranscription, mapRecognizeResponse, normalizeAudioInput };
|
|
273
|
+
|
|
274
|
+
//# sourceMappingURL=transcription.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"transcription.js","names":[],"sources":["../../../src/adapters/transcription.ts"],"sourcesContent":["import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'\nimport { arrayBufferToBase64, generateId } from '@tanstack/ai-utils'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport {\n BYTEPLUS_VOICE_BASE_URL,\n bytePlusVoiceError,\n bytePlusVoiceHeaders,\n getBytePlusVoiceApiKeyFromEnv,\n readJsonBody,\n withBytePlusVoiceDefaults,\n} from '../utils/client'\nimport {\n BYTEPLUS_ASR_RESOURCE_HEADER,\n BYTEPLUS_ASR_RESOURCE_ID,\n} from '../audio/wire-types'\nimport type {\n TokenUsage,\n TranscriptionOptions,\n TranscriptionResult,\n TranscriptionSegment,\n TranscriptionWord,\n} from '@tanstack/ai'\nimport type { InternalLogger } from '@tanstack/ai/adapter-internals'\nimport type { BytePlusVoiceConfig } from '../utils/client'\nimport type { BytePlusTranscriptionModel } from '../model-meta'\nimport type {\n BytePlusASRAudio,\n BytePlusASRRecognizeRequest,\n BytePlusASRRecognizeResponse,\n BytePlusASRUtterance,\n} from '../audio/wire-types'\nimport type { BytePlusTranscriptionProviderOptions } from '../audio/transcription-provider-options'\n\n/** Path of the synchronous (\"flash\") Seed ASR endpoint. */\nconst RECOGNIZE_FLASH_PATH = '/api/v3/auc/bigmodel/recognize/flash'\n\n/**\n * BytePlus-specific extension of `TranscriptionWord` carrying the per-word\n * confidence Seed ASR returns. The cross-provider contract has no field for\n * it, so callers who want it narrow the array — the same pattern the Grok\n * adapter uses:\n *\n * ```ts\n * const words = result.words as Array<BytePlusTranscriptionWord> | undefined\n * ```\n */\nexport interface BytePlusTranscriptionWord extends TranscriptionWord {\n /** Model confidence for the word, when Seed ASR returns one. */\n confidence?: number\n}\n\n/** Default `user.uid` echoed into BytePlus' request logs. */\nconst DEFAULT_UID = 'tanstack-ai'\n\n/**\n * BytePlus Seed Speech transcription (ASR) adapter.\n *\n * Talks to `POST {baseURL}/api/v3/auc/bigmodel/recognize/flash` — the\n * synchronous \"flash\" endpoint, which returns the whole transcript in one\n * response rather than requiring a submit/poll cycle. It accepts audio up to\n * 2 hours long or 100 MB, either as a publicly reachable URL or as base64\n * bytes.\n *\n * Two BytePlus-specific details:\n *\n * - The model is selected by the `X-Api-Resource-Id` header\n * (`volc.seedasr.auc_turbo`), not by a `model` field in the body. The\n * package's `seed-asr` model id exists to satisfy the SDK contract and to\n * give logs a stable value.\n * - Authentication uses `X-Api-Key` with the **Seed Speech** key, which is a\n * different key from `ARK_API_KEY`.\n *\n * All timings on the wire are milliseconds; they are converted to seconds to\n * match the cross-provider `TranscriptionResult`.\n *\n * @example\n * ```ts\n * const adapter = byteplusTranscription('seed-asr')\n * const result = await generateTranscription({\n * adapter,\n * audio: 'https://example.com/interview.mp3',\n * language: 'en-US',\n * })\n * ```\n */\nexport class BytePlusTranscriptionAdapter<\n TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,\n> extends BaseTranscriptionAdapter<\n TModel,\n BytePlusTranscriptionProviderOptions\n> {\n readonly name = 'byteplus' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n private readonly fetchImpl: typeof fetch\n\n constructor(model: TModel, config: BytePlusVoiceConfig) {\n super(model, config)\n const resolved = withBytePlusVoiceDefaults(config)\n this.apiKey = resolved.apiKey\n this.baseURL = resolved.baseURL ?? BYTEPLUS_VOICE_BASE_URL\n this.defaultHeaders = resolved.defaultHeaders ?? {}\n this.fetchImpl = resolved.fetch ?? globalThis.fetch.bind(globalThis)\n }\n\n async transcribe(\n options: TranscriptionOptions<BytePlusTranscriptionProviderOptions>,\n ): Promise<TranscriptionResult> {\n const {\n logger,\n model,\n audio,\n language,\n prompt,\n responseFormat,\n modelOptions,\n } = options\n\n logger.request(\n `activity=generateTranscription provider=byteplus model=${model}`,\n { provider: 'byteplus', model },\n )\n\n if (prompt) {\n logger.warn(\n 'BytePlus Seed ASR has no prompt-biasing field on the flash endpoint — the `prompt` option is ignored.',\n { provider: 'byteplus', model },\n )\n }\n\n // The flash endpoint answers with one JSON shape and offers no format\n // negotiation, so srt/vtt/text/verbose_json can't be honoured. `segments`\n // on the result carry the timings a caller would have wanted from srt/vtt.\n if (responseFormat !== undefined && responseFormat !== 'json') {\n logger.warn(\n `BytePlus Seed ASR always returns JSON — the requested responseFormat \"${responseFormat}\" is ignored. Build srt/vtt from result.segments if you need them.`,\n { provider: 'byteplus', model, responseFormat },\n )\n }\n\n try {\n const audioPayload = await normalizeAudioInput(\n audio,\n modelOptions?.audio_format,\n )\n const body = buildRecognizeRequestBody({\n audio: audioPayload,\n language,\n modelOptions,\n })\n\n const response = await this.fetchImpl(\n `${this.baseURL}${RECOGNIZE_FLASH_PATH}`,\n {\n method: 'POST',\n headers: bytePlusVoiceHeaders(this.apiKey, {\n ...this.defaultHeaders,\n [BYTEPLUS_ASR_RESOURCE_HEADER]: BYTEPLUS_ASR_RESOURCE_ID,\n }),\n body: JSON.stringify(body),\n },\n )\n\n const payload = await readJsonBody(response)\n\n if (!response.ok) {\n throw bytePlusVoiceError(response.status, payload, 'transcription')\n }\n\n const data = payload as BytePlusASRRecognizeResponse\n const text = data.result?.text ?? data.transcript\n\n // The flash endpoint can answer HTTP 200 while carrying the numeric\n // error envelope, so an absent transcript is a failure rather than an\n // empty result.\n if (typeof text !== 'string') {\n throw bytePlusVoiceError(response.status, payload, 'transcription')\n }\n\n // An empty string is well-formed, so it isn't an error — silence is a\n // legitimate transcription. But it is also what a 200-wrapped failure\n // looks like, so say so rather than handing back a successful, empty\n // result with no signal.\n if (text === '' && !hasUtterances(data)) {\n logger.warn(\n `byteplus: transcription returned an empty transcript with no ` +\n `utterances. This is a valid result for silent audio, and is also ` +\n `what a 200-wrapped failure looks like.`,\n { provider: this.name, model },\n )\n }\n\n // Seed ASR doesn't echo the language back, so report the one that was\n // actually sent — which is `modelOptions.language` when it overrode the\n // cross-provider hint.\n const requestedLanguage = modelOptions?.language ?? language\n\n return {\n id: generateId(this.name),\n model,\n ...mapRecognizeResponse(data, text, logger),\n ...(requestedLanguage !== undefined && { language: requestedLanguage }),\n }\n } catch (error) {\n logger.errors('byteplus.transcribe fatal', {\n error: toRunErrorPayload(error, 'byteplus.transcribe failed'),\n source: 'byteplus.transcribe',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the JSON body for `POST /api/v3/auc/bigmodel/recognize/flash`.\n *\n * `show_utterances` defaults to `true` so the response carries the\n * per-utterance breakdown that populates `segments` and `words`.\n */\nexport function buildRecognizeRequestBody(options: {\n audio: BytePlusASRAudio\n language: string | undefined\n modelOptions: BytePlusTranscriptionProviderOptions | undefined\n}): BytePlusASRRecognizeRequest {\n const { audio, language, modelOptions } = options\n\n const resolvedLanguage = modelOptions?.language ?? language\n\n return {\n user: { uid: modelOptions?.uid ?? DEFAULT_UID },\n audio,\n request: {\n model_name: modelOptions?.model_name ?? 'bigmodel',\n show_utterances: modelOptions?.show_utterances ?? true,\n ...(modelOptions?.enable_itn !== undefined && {\n enable_itn: modelOptions.enable_itn,\n }),\n ...(modelOptions?.enable_punc !== undefined && {\n enable_punc: modelOptions.enable_punc,\n }),\n ...(modelOptions?.enable_ddc !== undefined && {\n enable_ddc: modelOptions.enable_ddc,\n }),\n ...(modelOptions?.enable_speaker_info !== undefined && {\n enable_speaker_info: modelOptions.enable_speaker_info,\n }),\n ...(resolvedLanguage !== undefined && { language: resolvedLanguage }),\n },\n }\n}\n\n/**\n * Turn a recognition response into the transcript-shaped half of a\n * `TranscriptionResult`. Wire timings are milliseconds; everything returned\n * here is seconds.\n */\nexport function mapRecognizeResponse(\n data: BytePlusASRRecognizeResponse,\n text: string,\n logger?: InternalLogger,\n): Omit<TranscriptionResult, 'id' | 'model'> {\n const utterances = data.result?.utterances ?? data.utterances ?? []\n // `id` numbers the segments we emit, not the utterances we were given, so\n // dropping an untimed utterance doesn't leave a hole in the sequence.\n const segments = utterances\n .flatMap((utterance) => toSegment(utterance))\n .map((segment, index) => ({ ...segment, id: index }))\n\n const rawWords = utterances.flatMap((utterance) => utterance.words ?? [])\n const words = rawWords.flatMap((word) => {\n if (\n typeof word.text !== 'string' ||\n typeof word.start_time !== 'number' ||\n typeof word.end_time !== 'number'\n ) {\n return []\n }\n const mapped: BytePlusTranscriptionWord = {\n word: word.text,\n start: msToSeconds(word.start_time),\n end: msToSeconds(word.end_time),\n }\n if (word.confidence !== undefined) mapped.confidence = word.confidence\n return [mapped]\n })\n\n // Untimed entries are dropped rather than emitted with NaN timings, but a\n // silent drop leaves the caller unable to tell \"the provider sent no\n // timings\" from \"the adapter discarded them\" — the two have very different\n // fixes, and a field rename upstream (e.g. `text` → `word`) would empty\n // these arrays without a single error anywhere.\n const droppedWords = rawWords.length - words.length\n if (droppedWords > 0) {\n logger?.warn(\n `byteplus: dropped ${droppedWords} of ${rawWords.length} word(s) with ` +\n `missing or non-numeric timings.`,\n { provider: 'byteplus' },\n )\n }\n const droppedSegments = utterances.length - segments.length\n if (droppedSegments > 0) {\n logger?.warn(\n `byteplus: dropped ${droppedSegments} of ${utterances.length} ` +\n `utterance(s) with missing or non-numeric timings.`,\n { provider: 'byteplus' },\n )\n }\n\n const durationMs = data.audio_info?.duration\n const duration =\n typeof durationMs === 'number' && durationMs > 0\n ? msToSeconds(durationMs)\n : undefined\n\n // Seed ASR is duration-billed and reports no token counts, so `usage`\n // carries only the audio length — the same shape the Grok and OpenAI\n // whisper paths use.\n const usage: TokenUsage | undefined =\n duration !== undefined\n ? {\n promptTokens: 0,\n completionTokens: 0,\n totalTokens: 0,\n durationSeconds: duration,\n }\n : undefined\n\n return {\n text,\n ...(duration !== undefined && { duration }),\n ...(segments.length > 0 && { segments }),\n ...(words.length > 0 && { words }),\n ...(usage !== undefined && { usage }),\n }\n}\n\n/**\n * Convert one utterance into a segment, or nothing when it carries no\n * timings. The `id` is a placeholder — the caller renumbers after filtering.\n */\n/**\n * True when the response carries at least one utterance, in either envelope\n * form. Used to tell \"silent audio\" from a 200-wrapped failure: a genuinely\n * empty transcript usually still arrives with no utterances, so the pairing is\n * a hint rather than proof — hence a warning rather than a throw.\n */\nfunction hasUtterances(data: BytePlusASRRecognizeResponse): boolean {\n return (data.result?.utterances ?? data.utterances ?? []).length > 0\n}\n\nfunction toSegment(\n utterance: BytePlusASRUtterance,\n): Array<TranscriptionSegment> {\n if (\n typeof utterance.start_time !== 'number' ||\n typeof utterance.end_time !== 'number'\n ) {\n return []\n }\n const speaker = utterance.additions?.speaker\n return [\n {\n id: 0,\n start: msToSeconds(utterance.start_time),\n end: msToSeconds(utterance.end_time),\n text: utterance.text ?? '',\n ...(speaker !== undefined && { speaker }),\n },\n ]\n}\n\n/**\n * **Must verify when the Seed Speech key lands.** Every timing this adapter\n * reads — `audio_info.duration`, and each utterance's and word's\n * `start_time` / `end_time` — is assumed to be milliseconds. That comes from\n * the Volcengine flash-recognition reference this endpoint derives from\n * (a 2.499 s clip reports `duration: 2499`), not from a BytePlus response we\n * have seen. If BytePlus reports seconds instead, every duration, segment and\n * word timing here is 1000× too small, and this is the only place to fix.\n */\nfunction msToSeconds(milliseconds: number): number {\n return milliseconds / 1000\n}\n\n/**\n * Turn the cross-provider `audio` input into the endpoint's `audio` block.\n *\n * URLs are passed through untouched — Seed ASR fetches them itself, which\n * avoids pulling large media through this process. Everything else is sent as\n * base64 `data`, with the container inferred from the input's MIME type or\n * filename when the caller didn't pin `audio_format`.\n */\nexport async function normalizeAudioInput(\n audio: TranscriptionOptions['audio'],\n formatHint: string | undefined,\n): Promise<BytePlusASRAudio> {\n const withFormat = (\n payload: BytePlusASRAudio,\n inferred?: string,\n ): BytePlusASRAudio => {\n const format = formatHint ?? inferred\n return format ? { ...payload, format } : payload\n }\n\n if (typeof audio === 'string') {\n if (/^https?:\\/\\//i.test(audio)) {\n return withFormat({ url: audio }, extensionOf(audio))\n }\n const dataUrl = /^data:([^;,]+)?(?:;[^,]*)*,(.*)$/s.exec(audio)\n if (dataUrl) {\n return withFormat({ data: dataUrl[2] ?? '' }, formatFromMime(dataUrl[1]))\n }\n // A bare string that is neither a URL nor a data URL is already base64.\n return withFormat({ data: audio })\n }\n\n if (audio instanceof ArrayBuffer) {\n return withFormat({ data: arrayBufferToBase64(audio) })\n }\n\n const data = arrayBufferToBase64(await audio.arrayBuffer())\n const inferred =\n ('name' in audio && typeof audio.name === 'string'\n ? extensionOf(audio.name)\n : undefined) ?? formatFromMime(audio.type)\n return withFormat({ data }, inferred)\n}\n\nfunction extensionOf(pathOrName: string): string | undefined {\n const withoutQuery = pathOrName.split(/[?#]/)[0] ?? ''\n const match = /\\.([a-z0-9]+)$/i.exec(withoutQuery)\n return match?.[1]?.toLowerCase()\n}\n\nfunction formatFromMime(mime: string | undefined): string | undefined {\n if (!mime || !mime.startsWith('audio/')) return undefined\n const subtype = mime.slice('audio/'.length).toLowerCase()\n if (subtype === 'mpeg') return 'mp3'\n if (subtype === 'x-wav' || subtype === 'wave') return 'wav'\n return subtype.replace(/^x-/, '')\n}\n\n/**\n * Creates a BytePlus Seed Speech transcription adapter with an explicit API\n * key.\n *\n * The key is the **Seed Speech** key, not the Ark key used by the chat, image\n * and video adapters.\n */\nexport function createBytePlusTranscription<\n TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,\n>(\n model: TModel,\n apiKey: string,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTranscriptionAdapter<TModel> {\n return new BytePlusTranscriptionAdapter(model, { ...config, apiKey })\n}\n\n/**\n * Creates a BytePlus Seed Speech transcription adapter, reading the API key\n * from `BYTEPLUS_VOICE_API_KEY`.\n *\n * @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.\n */\nexport function byteplusTranscription<\n TModel extends BytePlusTranscriptionModel = BytePlusTranscriptionModel,\n>(\n model: TModel,\n config?: Omit<BytePlusVoiceConfig, 'apiKey'>,\n): BytePlusTranscriptionAdapter<TModel> {\n return createBytePlusTranscription(\n model,\n getBytePlusVoiceApiKeyFromEnv(),\n config,\n )\n}\n"],"mappings":";;;;;;;AAkCA,IAAM,uBAAuB;;AAkB7B,IAAM,cAAc;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAiCpB,IAAa,+BAAb,cAEU,yBAGR;CACA,OAAgB;CAEhB;CACA;CACA;CACA;CAEA,YAAY,OAAe,QAA6B;EACtD,MAAM,OAAO,MAAM;EACnB,MAAM,WAAW,0BAA0B,MAAM;EACjD,KAAK,SAAS,SAAS;EACvB,KAAK,UAAU,SAAS,WAAA;EACxB,KAAK,iBAAiB,SAAS,kBAAkB,CAAC;EAClD,KAAK,YAAY,SAAS,SAAS,WAAW,MAAM,KAAK,UAAU;CACrE;CAEA,MAAM,WACJ,SAC8B;EAC9B,MAAM,EACJ,QACA,OACA,OACA,UACA,QACA,gBACA,iBACE;EAEJ,OAAO,QACL,0DAA0D,SAC1D;GAAE,UAAU;GAAY;EAAM,CAChC;EAEA,IAAI,QACF,OAAO,KACL,yGACA;GAAE,UAAU;GAAY;EAAM,CAChC;EAMF,IAAI,mBAAmB,KAAA,KAAa,mBAAmB,QACrD,OAAO,KACL,yEAAyE,eAAe,qEACxF;GAAE,UAAU;GAAY;GAAO;EAAe,CAChD;EAGF,IAAI;GAKF,MAAM,OAAO,0BAA0B;IACrC,OAAO,MALkB,oBACzB,OACA,cAAc,YAChB;IAGE;IACA;GACF,CAAC;GAED,MAAM,WAAW,MAAM,KAAK,UAC1B,GAAG,KAAK,UAAU,wBAClB;IACE,QAAQ;IACR,SAAS,qBAAqB,KAAK,QAAQ;KACzC,GAAG,KAAK;MACP,+BAA+B;IAClC,CAAC;IACD,MAAM,KAAK,UAAU,IAAI;GAC3B,CACF;GAEA,MAAM,UAAU,MAAM,aAAa,QAAQ;GAE3C,IAAI,CAAC,SAAS,IACZ,MAAM,mBAAmB,SAAS,QAAQ,SAAS,eAAe;GAGpE,MAAM,OAAO;GACb,MAAM,OAAO,KAAK,QAAQ,QAAQ,KAAK;GAKvC,IAAI,OAAO,SAAS,UAClB,MAAM,mBAAmB,SAAS,QAAQ,SAAS,eAAe;GAOpE,IAAI,SAAS,MAAM,CAAC,cAAc,IAAI,GACpC,OAAO,KACL,wKAGA;IAAE,UAAU,KAAK;IAAM;GAAM,CAC/B;GAMF,MAAM,oBAAoB,cAAc,YAAY;GAEpD,OAAO;IACL,IAAI,WAAW,KAAK,IAAI;IACxB;IACA,GAAG,qBAAqB,MAAM,MAAM,MAAM;IAC1C,GAAI,sBAAsB,KAAA,KAAa,EAAE,UAAU,kBAAkB;GACvE;EACF,SAAS,OAAO;GACd,OAAO,OAAO,6BAA6B;IACzC,OAAO,kBAAkB,OAAO,4BAA4B;IAC5D,QAAQ;GACV,CAAC;GACD,MAAM;EACR;CACF;AACF;;;;;;;AAQA,SAAgB,0BAA0B,SAIV;CAC9B,MAAM,EAAE,OAAO,UAAU,iBAAiB;CAE1C,MAAM,mBAAmB,cAAc,YAAY;CAEnD,OAAO;EACL,MAAM,EAAE,KAAK,cAAc,OAAO,YAAY;EAC9C;EACA,SAAS;GACP,YAAY,cAAc,cAAc;GACxC,iBAAiB,cAAc,mBAAmB;GAClD,GAAI,cAAc,eAAe,KAAA,KAAa,EAC5C,YAAY,aAAa,WAC3B;GACA,GAAI,cAAc,gBAAgB,KAAA,KAAa,EAC7C,aAAa,aAAa,YAC5B;GACA,GAAI,cAAc,eAAe,KAAA,KAAa,EAC5C,YAAY,aAAa,WAC3B;GACA,GAAI,cAAc,wBAAwB,KAAA,KAAa,EACrD,qBAAqB,aAAa,oBACpC;GACA,GAAI,qBAAqB,KAAA,KAAa,EAAE,UAAU,iBAAiB;EACrE;CACF;AACF;;;;;;AAOA,SAAgB,qBACd,MACA,MACA,QAC2C;CAC3C,MAAM,aAAa,KAAK,QAAQ,cAAc,KAAK,cAAc,CAAC;CAGlE,MAAM,WAAW,WACd,SAAS,cAAc,UAAU,SAAS,CAAC,CAAC,CAC5C,KAAK,SAAS,WAAW;EAAE,GAAG;EAAS,IAAI;CAAM,EAAE;CAEtD,MAAM,WAAW,WAAW,SAAS,cAAc,UAAU,SAAS,CAAC,CAAC;CACxE,MAAM,QAAQ,SAAS,SAAS,SAAS;EACvC,IACE,OAAO,KAAK,SAAS,YACrB,OAAO,KAAK,eAAe,YAC3B,OAAO,KAAK,aAAa,UAEzB,OAAO,CAAC;EAEV,MAAM,SAAoC;GACxC,MAAM,KAAK;GACX,OAAO,YAAY,KAAK,UAAU;GAClC,KAAK,YAAY,KAAK,QAAQ;EAChC;EACA,IAAI,KAAK,eAAe,KAAA,GAAW,OAAO,aAAa,KAAK;EAC5D,OAAO,CAAC,MAAM;CAChB,CAAC;CAOD,MAAM,eAAe,SAAS,SAAS,MAAM;CAC7C,IAAI,eAAe,GACjB,QAAQ,KACN,qBAAqB,aAAa,MAAM,SAAS,OAAO,gDAExD,EAAE,UAAU,WAAW,CACzB;CAEF,MAAM,kBAAkB,WAAW,SAAS,SAAS;CACrD,IAAI,kBAAkB,GACpB,QAAQ,KACN,qBAAqB,gBAAgB,MAAM,WAAW,OAAO,qDAE7D,EAAE,UAAU,WAAW,CACzB;CAGF,MAAM,aAAa,KAAK,YAAY;CACpC,MAAM,WACJ,OAAO,eAAe,YAAY,aAAa,IAC3C,YAAY,UAAU,IACtB,KAAA;CAKN,MAAM,QACJ,aAAa,KAAA,IACT;EACE,cAAc;EACd,kBAAkB;EAClB,aAAa;EACb,iBAAiB;CACnB,IACA,KAAA;CAEN,OAAO;EACL;EACA,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;EACzC,GAAI,SAAS,SAAS,KAAK,EAAE,SAAS;EACtC,GAAI,MAAM,SAAS,KAAK,EAAE,MAAM;EAChC,GAAI,UAAU,KAAA,KAAa,EAAE,MAAM;CACrC;AACF;;;;;;;;;;;AAYA,SAAS,cAAc,MAA6C;CAClE,QAAQ,KAAK,QAAQ,cAAc,KAAK,cAAc,CAAC,EAAA,CAAG,SAAS;AACrE;AAEA,SAAS,UACP,WAC6B;CAC7B,IACE,OAAO,UAAU,eAAe,YAChC,OAAO,UAAU,aAAa,UAE9B,OAAO,CAAC;CAEV,MAAM,UAAU,UAAU,WAAW;CACrC,OAAO,CACL;EACE,IAAI;EACJ,OAAO,YAAY,UAAU,UAAU;EACvC,KAAK,YAAY,UAAU,QAAQ;EACnC,MAAM,UAAU,QAAQ;EACxB,GAAI,YAAY,KAAA,KAAa,EAAE,QAAQ;CACzC,CACF;AACF;;;;;;;;;;AAWA,SAAS,YAAY,cAA8B;CACjD,OAAO,eAAe;AACxB;;;;;;;;;AAUA,eAAsB,oBACpB,OACA,YAC2B;CAC3B,MAAM,cACJ,SACA,aACqB;EACrB,MAAM,SAAS,cAAc;EAC7B,OAAO,SAAS;GAAE,GAAG;GAAS;EAAO,IAAI;CAC3C;CAEA,IAAI,OAAO,UAAU,UAAU;EAC7B,IAAI,gBAAgB,KAAK,KAAK,GAC5B,OAAO,WAAW,EAAE,KAAK,MAAM,GAAG,YAAY,KAAK,CAAC;EAEtD,MAAM,UAAU,oCAAoC,KAAK,KAAK;EAC9D,IAAI,SACF,OAAO,WAAW,EAAE,MAAM,QAAQ,MAAM,GAAG,GAAG,eAAe,QAAQ,EAAE,CAAC;EAG1E,OAAO,WAAW,EAAE,MAAM,MAAM,CAAC;CACnC;CAEA,IAAI,iBAAiB,aACnB,OAAO,WAAW,EAAE,MAAM,oBAAoB,KAAK,EAAE,CAAC;CAGxD,MAAM,OAAO,oBAAoB,MAAM,MAAM,YAAY,CAAC;CAC1D,MAAM,YACH,UAAU,SAAS,OAAO,MAAM,SAAS,WACtC,YAAY,MAAM,IAAI,IACtB,KAAA,MAAc,eAAe,MAAM,IAAI;CAC7C,OAAO,WAAW,EAAE,KAAK,GAAG,QAAQ;AACtC;AAEA,SAAS,YAAY,YAAwC;CAC3D,MAAM,eAAe,WAAW,MAAM,MAAM,CAAC,CAAC,MAAM;CAEpD,OADc,kBAAkB,KAAK,YAC9B,CAAA,GAAQ,EAAE,EAAE,YAAY;AACjC;AAEA,SAAS,eAAe,MAA8C;CACpE,IAAI,CAAC,QAAQ,CAAC,KAAK,WAAW,QAAQ,GAAG,OAAO,KAAA;CAChD,MAAM,UAAU,KAAK,MAAM,CAAe,CAAC,CAAC,YAAY;CACxD,IAAI,YAAY,QAAQ,OAAO;CAC/B,IAAI,YAAY,WAAW,YAAY,QAAQ,OAAO;CACtD,OAAO,QAAQ,QAAQ,OAAO,EAAE;AAClC;;;;;;;;AASA,SAAgB,4BAGd,OACA,QACA,QACsC;CACtC,OAAO,IAAI,6BAA6B,OAAO;EAAE,GAAG;EAAQ;CAAO,CAAC;AACtE;;;;;;;AAQA,SAAgB,sBAGd,OACA,QACsC;CACtC,OAAO,4BACL,OACA,8BAA8B,GAC9B,MACF;AACF"}
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
import { BaseTTSAdapter } from '@tanstack/ai/adapters';
|
|
2
|
+
import { TTSOptions } from '@tanstack/ai';
|
|
3
|
+
import { InternalLogger } from '@tanstack/ai/adapter-internals';
|
|
4
|
+
import { BytePlusVoiceConfig } from '../utils/client.js';
|
|
5
|
+
import { BytePlusTTSModel } from '../model-meta.js';
|
|
6
|
+
import { BytePlusTTSAudioFormat, BytePlusTTSCreateRequest } from '../audio/wire-types.js';
|
|
7
|
+
import { BytePlusTTSProviderOptions, BytePlusTTSResult } from '../audio/tts-provider-options.js';
|
|
8
|
+
/**
|
|
9
|
+
* Voice used when neither `TTSOptions.voice` nor `modelOptions.speaker` is
|
|
10
|
+
* set — the English female "Stokie" voice from the TTS 2.0 generation.
|
|
11
|
+
*/
|
|
12
|
+
export declare const BYTEPLUS_DEFAULT_TTS_SPEAKER = "en_female_stokie_uranus_bigtts";
|
|
13
|
+
/**
|
|
14
|
+
* Hard cap on a single Seed Speech synthesis, in seconds. Longer scripts must
|
|
15
|
+
* be split across calls and stitched client-side.
|
|
16
|
+
*
|
|
17
|
+
* The cap applies to the *pre-rate* length the service bills on — the
|
|
18
|
+
* `original_duration` it returns. The delivered clip can run longer than this
|
|
19
|
+
* when `speech_rate` slows it down.
|
|
20
|
+
*/
|
|
21
|
+
export declare const BYTEPLUS_TTS_MAX_OUTPUT_SECONDS = 120;
|
|
22
|
+
/**
|
|
23
|
+
* BytePlus Seed Speech text-to-speech adapter.
|
|
24
|
+
*
|
|
25
|
+
* Talks to `POST {baseURL}/api/v3/tts/create` on the Seed Speech voice host.
|
|
26
|
+
* Two things differ from the Ark-hosted adapters in this package:
|
|
27
|
+
*
|
|
28
|
+
* - **Separate API key.** Seed Speech authenticates with `X-Api-Key` and its
|
|
29
|
+
* own key (`BYTEPLUS_VOICE_API_KEY`). An `ARK_API_KEY` sent here is
|
|
30
|
+
* rejected with `45000010 Invalid X-Api-Key`.
|
|
31
|
+
* - **120 s output cap.** A single call synthesises at most
|
|
32
|
+
* {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} seconds of audio; longer scripts
|
|
33
|
+
* have to be split and stitched client-side. The cap is measured before
|
|
34
|
+
* `speech_rate` is applied, so a slowed clip can play for longer than that
|
|
35
|
+
* — `result.duration` is the delivered length and `result.originalDuration`
|
|
36
|
+
* is the metered one.
|
|
37
|
+
*
|
|
38
|
+
* Streaming synthesis (`tts/unidirectional` and the WebSocket API) is not
|
|
39
|
+
* covered by this adapter — note that endpoint spells the text field `text`,
|
|
40
|
+
* while this one uses `text_prompt`.
|
|
41
|
+
*
|
|
42
|
+
* @example
|
|
43
|
+
* ```ts
|
|
44
|
+
* const adapter = byteplusSpeech('seed-audio-1.0')
|
|
45
|
+
* const result = await generateSpeech({
|
|
46
|
+
* adapter,
|
|
47
|
+
* text: 'welcome to the guitar store',
|
|
48
|
+
* voice: 'en_female_stokie_uranus_bigtts',
|
|
49
|
+
* format: 'mp3',
|
|
50
|
+
* })
|
|
51
|
+
* ```
|
|
52
|
+
*/
|
|
53
|
+
export declare class BytePlusTTSAdapter<TModel extends BytePlusTTSModel = BytePlusTTSModel> extends BaseTTSAdapter<TModel, BytePlusTTSProviderOptions> {
|
|
54
|
+
readonly name: "byteplus";
|
|
55
|
+
private readonly apiKey;
|
|
56
|
+
private readonly baseURL;
|
|
57
|
+
private readonly defaultHeaders;
|
|
58
|
+
private readonly fetchImpl;
|
|
59
|
+
constructor(model: TModel, config: BytePlusVoiceConfig);
|
|
60
|
+
generateSpeech(options: TTSOptions<BytePlusTTSProviderOptions>): Promise<BytePlusTTSResult>;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Build the JSON body for `POST /api/v3/tts/create`, resolving the voice,
|
|
64
|
+
* output format and rate fields in one place.
|
|
65
|
+
*
|
|
66
|
+
* Returns the request `body`, the resolved `audioFormat` and the
|
|
67
|
+
* `sampleRate`, which the caller reports on the result and turns into a
|
|
68
|
+
* `contentType`.
|
|
69
|
+
*/
|
|
70
|
+
export declare function buildTTSRequestBody(options: {
|
|
71
|
+
model: string;
|
|
72
|
+
text: string;
|
|
73
|
+
voice: string | undefined;
|
|
74
|
+
format: TTSOptions['format'] | undefined;
|
|
75
|
+
speed: number | undefined;
|
|
76
|
+
modelOptions: BytePlusTTSProviderOptions | undefined;
|
|
77
|
+
logger: InternalLogger;
|
|
78
|
+
}): {
|
|
79
|
+
body: BytePlusTTSCreateRequest;
|
|
80
|
+
audioFormat: BytePlusTTSAudioFormat;
|
|
81
|
+
sampleRate: number;
|
|
82
|
+
};
|
|
83
|
+
/**
|
|
84
|
+
* Convert the cross-provider `TTSOptions.speed` multiplier into Seed Speech's
|
|
85
|
+
* `speech_rate` percentage.
|
|
86
|
+
*
|
|
87
|
+
* ```
|
|
88
|
+
* speech_rate = clamp(round((speed - 1) * 100), -50, 100)
|
|
89
|
+
* ```
|
|
90
|
+
*
|
|
91
|
+
* so `0.5 → -50`, `1.0 → 0`, `1.5 → 50`, `2.0 → 100`.
|
|
92
|
+
*
|
|
93
|
+
* The range `-50`..`100` and both multiplier anchors (`-50` = 0.5×,
|
|
94
|
+
* `100` = 2×) are documented, and hold on both the `/tts/create` and
|
|
95
|
+
* `/tts/unidirectional` endpoints.
|
|
96
|
+
*
|
|
97
|
+
* `TTSOptions.speed` spans a wider 0.25×–4× than the endpoint supports, so
|
|
98
|
+
* anything outside 0.5×–2× clamps (and warns) rather than erroring.
|
|
99
|
+
*/
|
|
100
|
+
export declare function toSpeechRate(speed: number, logger?: InternalLogger): number;
|
|
101
|
+
/**
|
|
102
|
+
* MIME type for a Seed Speech output format.
|
|
103
|
+
*
|
|
104
|
+
* `pcm` is raw little-endian 16-bit samples, so its media type has to carry
|
|
105
|
+
* the sample rate (RFC 3551/3555). The adapter always resolves a rate, so one
|
|
106
|
+
* is always available here.
|
|
107
|
+
*/
|
|
108
|
+
export declare function getContentType(format: BytePlusTTSAudioFormat, sampleRate?: number): string;
|
|
109
|
+
/**
|
|
110
|
+
* Coerce a `duration` / `original_duration` field to a usable number of
|
|
111
|
+
* seconds.
|
|
112
|
+
*
|
|
113
|
+
* Both are documented as float **seconds**, so this only parses the string
|
|
114
|
+
* form and drops values that can't be a length (zero, negative, non-numeric).
|
|
115
|
+
* Note that `duration` may legitimately exceed
|
|
116
|
+
* {@link BYTEPLUS_TTS_MAX_OUTPUT_SECONDS} — a clip synthesised at
|
|
117
|
+
* `speech_rate: -50` runs at half speed, so up to 120 s of billed audio can
|
|
118
|
+
* be delivered as up to 240 s of playback. Do not "correct" a large value
|
|
119
|
+
* here; the cap applies to `original_duration`.
|
|
120
|
+
*
|
|
121
|
+
* The subtitle timings are the mixed-unit exception: those are milliseconds
|
|
122
|
+
* and are passed through untouched on {@link BytePlusTTSResult.subtitle}.
|
|
123
|
+
*/
|
|
124
|
+
export declare function toDurationSeconds(raw: number | string | undefined): number | undefined;
|
|
125
|
+
/**
|
|
126
|
+
* Creates a BytePlus Seed Speech TTS adapter with an explicit API key.
|
|
127
|
+
*
|
|
128
|
+
* The key is the **Seed Speech** key, not the Ark key used by the chat, image
|
|
129
|
+
* and video adapters.
|
|
130
|
+
*
|
|
131
|
+
* @example
|
|
132
|
+
* ```ts
|
|
133
|
+
* const adapter = createBytePlusSpeech('seed-audio-1.0', process.env.BYTEPLUS_VOICE_API_KEY!)
|
|
134
|
+
* ```
|
|
135
|
+
*/
|
|
136
|
+
export declare function createBytePlusSpeech<TModel extends BytePlusTTSModel = BytePlusTTSModel>(model: TModel, apiKey: string, config?: Omit<BytePlusVoiceConfig, 'apiKey'>): BytePlusTTSAdapter<TModel>;
|
|
137
|
+
/**
|
|
138
|
+
* Creates a BytePlus Seed Speech TTS adapter, reading the API key from
|
|
139
|
+
* `BYTEPLUS_VOICE_API_KEY`.
|
|
140
|
+
*
|
|
141
|
+
* @throws Error if `BYTEPLUS_VOICE_API_KEY` is not set.
|
|
142
|
+
*/
|
|
143
|
+
export declare function byteplusSpeech<TModel extends BytePlusTTSModel = BytePlusTTSModel>(model: TModel, config?: Omit<BytePlusVoiceConfig, 'apiKey'>): BytePlusTTSAdapter<TModel>;
|