@tanstack/ai-byteplus 0.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +202 -0
  3. package/dist/esm/adapters/image.d.ts +89 -0
  4. package/dist/esm/adapters/image.js +229 -0
  5. package/dist/esm/adapters/image.js.map +1 -0
  6. package/dist/esm/adapters/text.d.ts +163 -0
  7. package/dist/esm/adapters/text.js +347 -0
  8. package/dist/esm/adapters/text.js.map +1 -0
  9. package/dist/esm/adapters/transcription.d.ts +102 -0
  10. package/dist/esm/adapters/transcription.js +274 -0
  11. package/dist/esm/adapters/transcription.js.map +1 -0
  12. package/dist/esm/adapters/tts.d.ts +143 -0
  13. package/dist/esm/adapters/tts.js +307 -0
  14. package/dist/esm/adapters/tts.js.map +1 -0
  15. package/dist/esm/adapters/video.d.ts +182 -0
  16. package/dist/esm/adapters/video.js +442 -0
  17. package/dist/esm/adapters/video.js.map +1 -0
  18. package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
  19. package/dist/esm/audio/tts-provider-options.d.ts +114 -0
  20. package/dist/esm/audio/wire-types.d.ts +261 -0
  21. package/dist/esm/audio/wire-types.js +28 -0
  22. package/dist/esm/audio/wire-types.js.map +1 -0
  23. package/dist/esm/image/image-provider-options.d.ts +165 -0
  24. package/dist/esm/image/image-provider-options.js +134 -0
  25. package/dist/esm/image/image-provider-options.js.map +1 -0
  26. package/dist/esm/image/wire-types.d.ts +149 -0
  27. package/dist/esm/index.d.ts +25 -0
  28. package/dist/esm/index.js +11 -0
  29. package/dist/esm/message-types.d.ts +154 -0
  30. package/dist/esm/model-meta.d.ts +594 -0
  31. package/dist/esm/model-meta.js +619 -0
  32. package/dist/esm/model-meta.js.map +1 -0
  33. package/dist/esm/text/text-provider-options.d.ts +109 -0
  34. package/dist/esm/utils/client.d.ts +183 -0
  35. package/dist/esm/utils/client.js +253 -0
  36. package/dist/esm/utils/client.js.map +1 -0
  37. package/dist/esm/video/video-provider-options.d.ts +197 -0
  38. package/dist/esm/video/video-provider-options.js +191 -0
  39. package/dist/esm/video/video-provider-options.js.map +1 -0
  40. package/dist/esm/video/wire-types.d.ts +248 -0
  41. package/package.json +77 -0
  42. package/src/adapters/image.ts +409 -0
  43. package/src/adapters/text.ts +539 -0
  44. package/src/adapters/transcription.ts +479 -0
  45. package/src/adapters/tts.ts +447 -0
  46. package/src/adapters/video.ts +732 -0
  47. package/src/audio/transcription-provider-options.ts +46 -0
  48. package/src/audio/tts-provider-options.ts +122 -0
  49. package/src/audio/wire-types.ts +290 -0
  50. package/src/image/image-provider-options.ts +288 -0
  51. package/src/image/wire-types.ts +169 -0
  52. package/src/index.ts +222 -0
  53. package/src/message-types.ts +169 -0
  54. package/src/model-meta.ts +954 -0
  55. package/src/text/text-provider-options.ts +151 -0
  56. package/src/utils/client.ts +377 -0
  57. package/src/video/video-provider-options.ts +361 -0
  58. package/src/video/wire-types.ts +293 -0
@@ -0,0 +1,114 @@
1
+ import { BytePlusTTSAudioFormat, BytePlusTTSReference, BytePlusTTSSampleRate, BytePlusTTSSubtitle } from './wire-types.js';
2
+ import { TTSResult } from '@tanstack/ai';
3
+ /**
4
+ * Seed Speech voice identifier (`speaker` on the wire).
5
+ *
6
+ * Voice ids encode language, gender, character name and model generation:
7
+ * `en_female_stokie_uranus_bigtts` is the English female "Stokie" voice on
8
+ * TTS 2.0. The generation suffix matters when picking one:
9
+ *
10
+ * - `_uranus_bigtts` — TTS 2.0 voices (the current generation).
11
+ * - `_mars_bigtts` / `_moon_bigtts` — TTS 1.0 voices.
12
+ * - `*_emo_v2_*` — TTS 1.0 voices that additionally accept emotion tags.
13
+ *
14
+ * The full roster lives at
15
+ * https://docs.byteplus.com/en/docs/byteplusvoice/voicelist and changes far
16
+ * more often than this package ships, so the union stays open: any string is
17
+ * accepted, and the one listed id is the adapter's default.
18
+ */
19
+ export type BytePlusTTSVoice = 'en_female_stokie_uranus_bigtts' | (string & {});
20
+ /**
21
+ * Provider-specific options for BytePlus Seed Speech TTS
22
+ * (`POST /api/v3/tts/create`).
23
+ *
24
+ * These map 1:1 onto the wire fields so the BytePlus documentation stays
25
+ * useful; where a cross-provider `TTSOptions` field covers the same ground
26
+ * (`voice`, `format`, `speed`), the option here wins.
27
+ */
28
+ export interface BytePlusTTSProviderOptions {
29
+ /**
30
+ * Voice id. Overrides `TTSOptions.voice` when both are set. It is sent as
31
+ * the `speaker` of a single `references` entry.
32
+ */
33
+ speaker?: BytePlusTTSVoice;
34
+ /**
35
+ * Full `references` array, for voice cloning or image-referenced delivery.
36
+ * When set this replaces the entry the adapter would otherwise build from
37
+ * `speaker` / `TTSOptions.voice`, so include a `speaker` entry yourself if
38
+ * you still want a stock voice. Address audio references from the text with
39
+ * `@Audio1`..`@Audio3`.
40
+ */
41
+ references?: Array<BytePlusTTSReference>;
42
+ /**
43
+ * Output format. Overrides the mapping applied to `TTSOptions.format`,
44
+ * which is useful for `ogg_opus` and for pinning `pcm` explicitly.
45
+ */
46
+ format?: BytePlusTTSAudioFormat;
47
+ /**
48
+ * Output sample rate in Hz. Defaults to 24000 — the adapter always sends an
49
+ * explicit rate because the documented server default (40000) is not one of
50
+ * the values the endpoint accepts. For `pcm` output this is also what the
51
+ * returned `contentType` (`audio/L16;rate=…`) reports.
52
+ */
53
+ sample_rate?: BytePlusTTSSampleRate;
54
+ /**
55
+ * Pitch adjustment in the range `-12`..`12`, where `0` is the voice's
56
+ * natural pitch.
57
+ */
58
+ pitch_rate?: number;
59
+ /**
60
+ * Speaking rate in the range `-50`..`100` (`-50` = 0.5×, `0` = 1×,
61
+ * `100` = 2×). Overrides the value derived from `TTSOptions.speed`.
62
+ */
63
+ speech_rate?: number;
64
+ /**
65
+ * Loudness adjustment in the range `-50`..`100`, where `0` is the voice's
66
+ * natural level.
67
+ */
68
+ loudness_rate?: number;
69
+ /**
70
+ * Ask for sentence- and word-level timings alongside the audio. They are
71
+ * surfaced on {@link BytePlusTTSResult.subtitle}.
72
+ */
73
+ enable_subtitle?: boolean;
74
+ /**
75
+ * Watermark the generated audio. The field name is confirmed against the
76
+ * endpoint schema; the boolean type is assumed and unprobed.
77
+ */
78
+ watermark?: boolean;
79
+ }
80
+ /**
81
+ * BytePlus-specific extension of `TTSResult`.
82
+ *
83
+ * The cross-provider `TTSResult` has nowhere to put the subtitle timings or
84
+ * the temporary download URL, so callers who want them narrow the result:
85
+ *
86
+ * ```ts
87
+ * const result: BytePlusTTSResult = await generateSpeech({ adapter, text })
88
+ * for (const sentence of result.subtitle?.sentences ?? []) {
89
+ * console.log(sentence.text, sentence.start_time)
90
+ * }
91
+ * ```
92
+ */
93
+ export interface BytePlusTTSResult extends TTSResult {
94
+ /**
95
+ * Sentence and word timings, present only when
96
+ * `modelOptions.enable_subtitle` was set. Their `start_time` / `end_time`
97
+ * are **milliseconds**, even though `duration` and
98
+ * {@link BytePlusTTSResult.originalDuration} are seconds.
99
+ */
100
+ subtitle?: BytePlusTTSSubtitle;
101
+ /**
102
+ * Length of the audio in seconds *before* `speech_rate` was applied. This
103
+ * is what BytePlus bills on and what the 120 s cap applies to, so it is the
104
+ * number to meter against — `duration` reflects the delivered clip and can
105
+ * exceed 120 s when the speech is slowed down.
106
+ */
107
+ originalDuration?: number;
108
+ /**
109
+ * Temporary download URL for the same audio that `audio` carries as base64.
110
+ * **Expires roughly 2 hours after generation** — persist the bytes, not the
111
+ * link.
112
+ */
113
+ url?: string;
114
+ }
@@ -0,0 +1,261 @@
1
+ /**
2
+ * Minimal wire types for the BytePlus **Seed Speech** HTTP API (TTS + ASR).
3
+ *
4
+ * Seed Speech is a separate product from Ark: it lives on
5
+ * `voice.ap-southeast-1.bytepluses.com`, authenticates with `X-Api-Key`
6
+ * (a different key from `ARK_API_KEY`), and returns a flat numeric error
7
+ * envelope instead of Ark's OpenAI-shaped one.
8
+ *
9
+ * Only the fields the adapters read or write are modelled here — this is a
10
+ * hand-written subset, not a generated schema.
11
+ *
12
+ * Provenance:
13
+ * - Endpoints, auth header, format/rate ranges and the 120 s TTS output cap:
14
+ * BytePlus Seed Speech docs (`docs.byteplus.com/en/docs/byteplusvoice`),
15
+ * captured in the Phase 0 research notes.
16
+ * - Error envelope `{code, message}`: verified live — an Ark key sent as
17
+ * `X-Api-Key` returns HTTP 401 `{"code":45000010,"message":"Invalid X-Api-Key"}`.
18
+ * - ASR request/response shape (`user`/`audio`/`request` in, `audio_info` +
19
+ * `result.utterances` out, all timings in **milliseconds**): the Volcengine
20
+ * flash-recognition reference the BytePlus endpoint is derived from
21
+ * (`docs.volcengine.com/docs/6561/1631584`).
22
+ *
23
+ * No Seed Speech API key was available when these were written, so the TTS
24
+ * response fields are documented-but-unverified; the adapters parse them
25
+ * defensively rather than assuming they are always present.
26
+ */
27
+ /** Output container/codec accepted by `audio_config.format`. */
28
+ export type BytePlusTTSAudioFormat = 'wav' | 'mp3' | 'pcm' | 'ogg_opus';
29
+ /**
30
+ * Sample rates `audio_config.sample_rate` accepts.
31
+ *
32
+ * The docs also state a *default* of 40000, which is not one of the valid
33
+ * values — a documentation bug. The adapter therefore always sends an
34
+ * explicit rate rather than relying on the server default.
35
+ */
36
+ export declare const BYTEPLUS_TTS_SAMPLE_RATES: readonly [8000, 16000, 24000, 32000, 44100, 48000];
37
+ /** A sample rate `audio_config.sample_rate` accepts. */
38
+ export type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number];
39
+ /**
40
+ * One entry of the request's `references` array.
41
+ *
42
+ * This is where the voice lives: `speaker` names a stock voice, while
43
+ * `audio_url` / `audio_data` supply reference clips to clone (the `@Audio1`..
44
+ * `@Audio3` markers in `text_prompt` address them positionally). Exactly one
45
+ * of `speaker`, `audio_url` or `audio_data` may be set per entry; at most 3
46
+ * audio references (30 s / 10 MB each) and 1 image reference (10 MB), and
47
+ * image references are mutually exclusive with audio ones.
48
+ *
49
+ * **Member object shape is unresolved — must live-probe when the Seed Speech
50
+ * key lands.** The docs list the member fields flat (`speaker | audio_data |
51
+ * audio_url | image_data | image_url`) without a worked example, so whether
52
+ * the server wants a flat `{ speaker }` or a discriminated `{ type, ... }`
53
+ * could not be settled. The adapter sends the flat reading — see
54
+ * `buildTTSRequestBody` in `../adapters/tts`.
55
+ */
56
+ export interface BytePlusTTSReference {
57
+ /** Stock voice id, e.g. `en_female_stokie_uranus_bigtts`. */
58
+ speaker?: string;
59
+ /** URL of a reference clip to clone (≤30 s, ≤10 MB). */
60
+ audio_url?: string;
61
+ /** Base64 reference clip to clone (≤30 s, ≤10 MB). */
62
+ audio_data?: string;
63
+ /** URL of a reference image (≤10 MB). Mutually exclusive with audio refs. */
64
+ image_url?: string;
65
+ /** Base64 reference image (≤10 MB). Mutually exclusive with audio refs. */
66
+ image_data?: string;
67
+ }
68
+ /**
69
+ * `audio_config` block of a TTS request — exactly six fields.
70
+ *
71
+ * The three `*_rate` fields are integer percentages relative to the voice's
72
+ * neutral delivery, not multipliers.
73
+ */
74
+ export interface BytePlusTTSAudioConfig {
75
+ /** Output format. Defaults to `wav` server-side. */
76
+ format?: BytePlusTTSAudioFormat;
77
+ /**
78
+ * Output sample rate in Hz. See {@link BYTEPLUS_TTS_SAMPLE_RATES} for the
79
+ * valid values and for why the adapter always sends one.
80
+ */
81
+ sample_rate?: number;
82
+ /** Speaking rate, `-50`..`100`. `-50` = 0.5×, `0` = 1×, `100` = 2×. */
83
+ speech_rate?: number;
84
+ /** Loudness adjustment, `-50`..`100`. `0` is the voice's natural level. */
85
+ loudness_rate?: number;
86
+ /** Pitch adjustment, `-12`..`12`. `0` is the voice's natural pitch. */
87
+ pitch_rate?: number;
88
+ /** Emit sentence and word timings in the response. Defaults to `false`. */
89
+ enable_subtitle?: boolean;
90
+ }
91
+ /** Request body for `POST /api/v3/tts/create` — exactly five fields. */
92
+ export interface BytePlusTTSCreateRequest {
93
+ /** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */
94
+ model: string;
95
+ /**
96
+ * The text to speak (≤3000 chars). Dual-purpose: either literal text or a
97
+ * natural-language description of the delivery, and the place the
98
+ * `@Audio1`..`@Audio3` reference markers go.
99
+ *
100
+ * **`text_prompt` is correct — do not "fix" this to `text`.** This endpoint
101
+ * has no request-side `text` field. The `text` spelling belongs to the
102
+ * *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits under
103
+ * `req_params.text`. Confirmed against
104
+ * docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01, which lists this
105
+ * body as exactly `model`, `text_prompt`, `references`, `audio_config`,
106
+ * `watermark`.
107
+ */
108
+ text_prompt: string;
109
+ /** Voice selection and cloning references. See {@link BytePlusTTSReference}. */
110
+ references?: Array<BytePlusTTSReference>;
111
+ audio_config?: BytePlusTTSAudioConfig;
112
+ /**
113
+ * Watermark the generated audio. The field name is confirmed; the boolean
114
+ * type is assumed by analogy with Seedream's `watermark` and unprobed.
115
+ */
116
+ watermark?: boolean;
117
+ }
118
+ /**
119
+ * One timed entry of a TTS subtitle track.
120
+ *
121
+ * **Times are milliseconds** — unlike the response's `duration` fields, which
122
+ * are seconds. The endpoint genuinely mixes units.
123
+ */
124
+ export interface BytePlusTTSSubtitleEntry {
125
+ text?: string;
126
+ start_time?: number;
127
+ end_time?: number;
128
+ }
129
+ /** Sentence- and word-level timings returned when `enable_subtitle` is set. */
130
+ export interface BytePlusTTSSubtitle {
131
+ sentences?: Array<BytePlusTTSSubtitleEntry>;
132
+ words?: Array<BytePlusTTSSubtitleEntry>;
133
+ }
134
+ /** Response body for `POST /api/v3/tts/create`. */
135
+ export interface BytePlusTTSCreateResponse {
136
+ /**
137
+ * Status code — `0` on success, a flat error code otherwise.
138
+ *
139
+ * Typed as `number | string` because only the *error* envelope was verified
140
+ * live (HTTP 401 `{"code":45000010,…}`); the success envelope's shape is
141
+ * docs-derived and no voice key was available to confirm it. Treating a
142
+ * string code as "not a failure" would return a failed 200 as success, so
143
+ * the adapter coerces before comparing — see `isZeroCode` in
144
+ * `adapters/tts.ts`.
145
+ */
146
+ code?: number | string;
147
+ message?: string;
148
+ /** Base64-encoded audio in the requested `audio_config.format`. */
149
+ audio?: string;
150
+ /**
151
+ * Length of the delivered audio in **seconds** (float), after `speech_rate`
152
+ * is applied. This can legitimately exceed 120 when the clip is slowed
153
+ * down — the 120 s cap applies to {@link BytePlusTTSCreateResponse.original_duration}.
154
+ */
155
+ duration?: number | string;
156
+ /**
157
+ * Length in **seconds** (float) before rate adjustment. This is the billing
158
+ * basis and is capped at 120.
159
+ */
160
+ original_duration?: number | string;
161
+ /** Temporary download URL for the same audio. Expires after ~2 hours. */
162
+ url?: string;
163
+ /** Sentence and word timings, present when `enable_subtitle` was set. */
164
+ subtitle?: BytePlusTTSSubtitle;
165
+ }
166
+ /**
167
+ * Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo
168
+ * model. The flash endpoint takes no `model` field in its body — the model is
169
+ * chosen entirely by this header.
170
+ */
171
+ export declare const BYTEPLUS_ASR_RESOURCE_ID = "volc.seedasr.auc_turbo";
172
+ /** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */
173
+ export declare const BYTEPLUS_ASR_RESOURCE_HEADER = "X-Api-Resource-Id";
174
+ /**
175
+ * Audio input. Exactly one of `url` or `data` is sent — the endpoint accepts
176
+ * files up to 2 hours long / 100 MB.
177
+ */
178
+ export interface BytePlusASRAudio {
179
+ /** Publicly reachable URL of the audio file. */
180
+ url?: string;
181
+ /** Base64-encoded audio bytes. */
182
+ data?: string;
183
+ /** Container hint, e.g. `mp3`, `wav`, `ogg`. */
184
+ format?: string;
185
+ }
186
+ /** `request` block of a recognition call. */
187
+ export interface BytePlusASRRequestOptions {
188
+ /** Recognition model family. Defaults to `bigmodel`. */
189
+ model_name?: string;
190
+ /** Inverse text normalisation (spoken numbers → digits). */
191
+ enable_itn?: boolean;
192
+ /** Insert punctuation. */
193
+ enable_punc?: boolean;
194
+ /** Disfluency removal ("um", repeated words). */
195
+ enable_ddc?: boolean;
196
+ /** Attach per-utterance speaker labels. */
197
+ enable_speaker_info?: boolean;
198
+ /** Return the `utterances` breakdown as well as the flat transcript. */
199
+ show_utterances?: boolean;
200
+ /** Spoken language hint, e.g. `en-US`. */
201
+ language?: string;
202
+ }
203
+ /** Request body for `POST /api/v3/auc/bigmodel/recognize/flash`. */
204
+ export interface BytePlusASRRecognizeRequest {
205
+ user?: {
206
+ uid?: string;
207
+ };
208
+ audio: BytePlusASRAudio;
209
+ request?: BytePlusASRRequestOptions;
210
+ }
211
+ /** One recognised word. `start_time` / `end_time` are milliseconds. */
212
+ export interface BytePlusASRWord {
213
+ text?: string;
214
+ start_time?: number;
215
+ end_time?: number;
216
+ confidence?: number;
217
+ }
218
+ /** One recognised utterance. `start_time` / `end_time` are milliseconds. */
219
+ export interface BytePlusASRUtterance {
220
+ text?: string;
221
+ start_time?: number;
222
+ end_time?: number;
223
+ words?: Array<BytePlusASRWord>;
224
+ /**
225
+ * Extra per-utterance annotations. Speaker labels arrive here when
226
+ * `enable_speaker_info` is set; the exact key is read defensively because it
227
+ * could not be confirmed against a live response.
228
+ */
229
+ additions?: Record<string, string>;
230
+ }
231
+ export interface BytePlusASRResult {
232
+ text?: string;
233
+ utterances?: Array<BytePlusASRUtterance>;
234
+ }
235
+ /**
236
+ * Response body for `POST /api/v3/auc/bigmodel/recognize/flash`.
237
+ *
238
+ * The Volcengine-lineage wire shape nests everything under `result`; BytePlus'
239
+ * prose docs describe the same payload as "transcript + utterances", so the
240
+ * flat spelling is tolerated as a fallback.
241
+ */
242
+ export interface BytePlusASRRecognizeResponse {
243
+ /** `duration` is the audio length in **milliseconds**. */
244
+ audio_info?: {
245
+ duration?: number;
246
+ };
247
+ result?: BytePlusASRResult;
248
+ /** Flat alias for `result.text`. */
249
+ transcript?: string;
250
+ /** Flat alias for `result.utterances`. */
251
+ utterances?: Array<BytePlusASRUtterance>;
252
+ }
253
+ /**
254
+ * Seed Speech error envelope: a flat numeric `code` plus a `message`, e.g.
255
+ * `{"code": 45000010, "message": "Invalid X-Api-Key"}` (verified live on a
256
+ * 401). Format it with `bytePlusVoiceError` from `../utils/client`.
257
+ */
258
+ export interface BytePlusVoiceErrorBody {
259
+ code?: number;
260
+ message?: string;
261
+ }
@@ -0,0 +1,28 @@
1
+ //#region src/audio/wire-types.ts
2
+ /**
3
+ * Sample rates `audio_config.sample_rate` accepts.
4
+ *
5
+ * The docs also state a *default* of 40000, which is not one of the valid
6
+ * values — a documentation bug. The adapter therefore always sends an
7
+ * explicit rate rather than relying on the server default.
8
+ */
9
+ var BYTEPLUS_TTS_SAMPLE_RATES = [
10
+ 8e3,
11
+ 16e3,
12
+ 24e3,
13
+ 32e3,
14
+ 44100,
15
+ 48e3
16
+ ];
17
+ /**
18
+ * Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo
19
+ * model. The flash endpoint takes no `model` field in its body — the model is
20
+ * chosen entirely by this header.
21
+ */
22
+ var BYTEPLUS_ASR_RESOURCE_ID = "volc.seedasr.auc_turbo";
23
+ /** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */
24
+ var BYTEPLUS_ASR_RESOURCE_HEADER = "X-Api-Resource-Id";
25
+ //#endregion
26
+ export { BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_TTS_SAMPLE_RATES };
27
+
28
+ //# sourceMappingURL=wire-types.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"wire-types.js","names":[],"sources":["../../../src/audio/wire-types.ts"],"sourcesContent":["/**\n * Minimal wire types for the BytePlus **Seed Speech** HTTP API (TTS + ASR).\n *\n * Seed Speech is a separate product from Ark: it lives on\n * `voice.ap-southeast-1.bytepluses.com`, authenticates with `X-Api-Key`\n * (a different key from `ARK_API_KEY`), and returns a flat numeric error\n * envelope instead of Ark's OpenAI-shaped one.\n *\n * Only the fields the adapters read or write are modelled here — this is a\n * hand-written subset, not a generated schema.\n *\n * Provenance:\n * - Endpoints, auth header, format/rate ranges and the 120 s TTS output cap:\n * BytePlus Seed Speech docs (`docs.byteplus.com/en/docs/byteplusvoice`),\n * captured in the Phase 0 research notes.\n * - Error envelope `{code, message}`: verified live — an Ark key sent as\n * `X-Api-Key` returns HTTP 401 `{\"code\":45000010,\"message\":\"Invalid X-Api-Key\"}`.\n * - ASR request/response shape (`user`/`audio`/`request` in, `audio_info` +\n * `result.utterances` out, all timings in **milliseconds**): the Volcengine\n * flash-recognition reference the BytePlus endpoint is derived from\n * (`docs.volcengine.com/docs/6561/1631584`).\n *\n * No Seed Speech API key was available when these were written, so the TTS\n * response fields are documented-but-unverified; the adapters parse them\n * defensively rather than assuming they are always present.\n */\n\n// ============================================================================\n// TTS — POST /api/v3/tts/create\n// ============================================================================\n\n/** Output container/codec accepted by `audio_config.format`. */\nexport type BytePlusTTSAudioFormat = 'wav' | 'mp3' | 'pcm' | 'ogg_opus'\n\n/**\n * Sample rates `audio_config.sample_rate` accepts.\n *\n * The docs also state a *default* of 40000, which is not one of the valid\n * values — a documentation bug. The adapter therefore always sends an\n * explicit rate rather than relying on the server default.\n */\nexport const BYTEPLUS_TTS_SAMPLE_RATES = [\n 8000, 16000, 24000, 32000, 44100, 48000,\n] as const\n\n/** A sample rate `audio_config.sample_rate` accepts. */\nexport type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number]\n\n/**\n * One entry of the request's `references` array.\n *\n * This is where the voice lives: `speaker` names a stock voice, while\n * `audio_url` / `audio_data` supply reference clips to clone (the `@Audio1`..\n * `@Audio3` markers in `text_prompt` address them positionally). Exactly one\n * of `speaker`, `audio_url` or `audio_data` may be set per entry; at most 3\n * audio references (30 s / 10 MB each) and 1 image reference (10 MB), and\n * image references are mutually exclusive with audio ones.\n *\n * **Member object shape is unresolved — must live-probe when the Seed Speech\n * key lands.** The docs list the member fields flat (`speaker | audio_data |\n * audio_url | image_data | image_url`) without a worked example, so whether\n * the server wants a flat `{ speaker }` or a discriminated `{ type, ... }`\n * could not be settled. The adapter sends the flat reading — see\n * `buildTTSRequestBody` in `../adapters/tts`.\n */\nexport interface BytePlusTTSReference {\n /** Stock voice id, e.g. `en_female_stokie_uranus_bigtts`. */\n speaker?: string\n /** URL of a reference clip to clone (≤30 s, ≤10 MB). */\n audio_url?: string\n /** Base64 reference clip to clone (≤30 s, ≤10 MB). */\n audio_data?: string\n /** URL of a reference image (≤10 MB). Mutually exclusive with audio refs. */\n image_url?: string\n /** Base64 reference image (≤10 MB). Mutually exclusive with audio refs. */\n image_data?: string\n}\n\n/**\n * `audio_config` block of a TTS request — exactly six fields.\n *\n * The three `*_rate` fields are integer percentages relative to the voice's\n * neutral delivery, not multipliers.\n */\nexport interface BytePlusTTSAudioConfig {\n /** Output format. Defaults to `wav` server-side. */\n format?: BytePlusTTSAudioFormat\n /**\n * Output sample rate in Hz. See {@link BYTEPLUS_TTS_SAMPLE_RATES} for the\n * valid values and for why the adapter always sends one.\n */\n sample_rate?: number\n /** Speaking rate, `-50`..`100`. `-50` = 0.5×, `0` = 1×, `100` = 2×. */\n speech_rate?: number\n /** Loudness adjustment, `-50`..`100`. `0` is the voice's natural level. */\n loudness_rate?: number\n /** Pitch adjustment, `-12`..`12`. `0` is the voice's natural pitch. */\n pitch_rate?: number\n /** Emit sentence and word timings in the response. Defaults to `false`. */\n enable_subtitle?: boolean\n}\n\n/** Request body for `POST /api/v3/tts/create` — exactly five fields. */\nexport interface BytePlusTTSCreateRequest {\n /** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */\n model: string\n /**\n * The text to speak (≤3000 chars). Dual-purpose: either literal text or a\n * natural-language description of the delivery, and the place the\n * `@Audio1`..`@Audio3` reference markers go.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** This endpoint\n * has no request-side `text` field. The `text` spelling belongs to the\n * *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits under\n * `req_params.text`. Confirmed against\n * docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01, which lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`.\n */\n text_prompt: string\n /** Voice selection and cloning references. See {@link BytePlusTTSReference}. */\n references?: Array<BytePlusTTSReference>\n audio_config?: BytePlusTTSAudioConfig\n /**\n * Watermark the generated audio. The field name is confirmed; the boolean\n * type is assumed by analogy with Seedream's `watermark` and unprobed.\n */\n watermark?: boolean\n}\n\n/**\n * One timed entry of a TTS subtitle track.\n *\n * **Times are milliseconds** — unlike the response's `duration` fields, which\n * are seconds. The endpoint genuinely mixes units.\n */\nexport interface BytePlusTTSSubtitleEntry {\n text?: string\n start_time?: number\n end_time?: number\n}\n\n/** Sentence- and word-level timings returned when `enable_subtitle` is set. */\nexport interface BytePlusTTSSubtitle {\n sentences?: Array<BytePlusTTSSubtitleEntry>\n words?: Array<BytePlusTTSSubtitleEntry>\n}\n\n/** Response body for `POST /api/v3/tts/create`. */\nexport interface BytePlusTTSCreateResponse {\n /**\n * Status code — `0` on success, a flat error code otherwise.\n *\n * Typed as `number | string` because only the *error* envelope was verified\n * live (HTTP 401 `{\"code\":45000010,…}`); the success envelope's shape is\n * docs-derived and no voice key was available to confirm it. Treating a\n * string code as \"not a failure\" would return a failed 200 as success, so\n * the adapter coerces before comparing — see `isZeroCode` in\n * `adapters/tts.ts`.\n */\n code?: number | string\n message?: string\n /** Base64-encoded audio in the requested `audio_config.format`. */\n audio?: string\n /**\n * Length of the delivered audio in **seconds** (float), after `speech_rate`\n * is applied. This can legitimately exceed 120 when the clip is slowed\n * down — the 120 s cap applies to {@link BytePlusTTSCreateResponse.original_duration}.\n */\n duration?: number | string\n /**\n * Length in **seconds** (float) before rate adjustment. This is the billing\n * basis and is capped at 120.\n */\n original_duration?: number | string\n /** Temporary download URL for the same audio. Expires after ~2 hours. */\n url?: string\n /** Sentence and word timings, present when `enable_subtitle` was set. */\n subtitle?: BytePlusTTSSubtitle\n}\n\n// ============================================================================\n// ASR — POST /api/v3/auc/bigmodel/recognize/flash\n// ============================================================================\n\n/**\n * Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo\n * model. The flash endpoint takes no `model` field in its body — the model is\n * chosen entirely by this header.\n */\nexport const BYTEPLUS_ASR_RESOURCE_ID = 'volc.seedasr.auc_turbo'\n\n/** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */\nexport const BYTEPLUS_ASR_RESOURCE_HEADER = 'X-Api-Resource-Id'\n\n/**\n * Audio input. Exactly one of `url` or `data` is sent — the endpoint accepts\n * files up to 2 hours long / 100 MB.\n */\nexport interface BytePlusASRAudio {\n /** Publicly reachable URL of the audio file. */\n url?: string\n /** Base64-encoded audio bytes. */\n data?: string\n /** Container hint, e.g. `mp3`, `wav`, `ogg`. */\n format?: string\n}\n\n/** `request` block of a recognition call. */\nexport interface BytePlusASRRequestOptions {\n /** Recognition model family. Defaults to `bigmodel`. */\n model_name?: string\n /** Inverse text normalisation (spoken numbers → digits). */\n enable_itn?: boolean\n /** Insert punctuation. */\n enable_punc?: boolean\n /** Disfluency removal (\"um\", repeated words). */\n enable_ddc?: boolean\n /** Attach per-utterance speaker labels. */\n enable_speaker_info?: boolean\n /** Return the `utterances` breakdown as well as the flat transcript. */\n show_utterances?: boolean\n /** Spoken language hint, e.g. `en-US`. */\n language?: string\n}\n\n/** Request body for `POST /api/v3/auc/bigmodel/recognize/flash`. */\nexport interface BytePlusASRRecognizeRequest {\n user?: { uid?: string }\n audio: BytePlusASRAudio\n request?: BytePlusASRRequestOptions\n}\n\n/** One recognised word. `start_time` / `end_time` are milliseconds. */\nexport interface BytePlusASRWord {\n text?: string\n start_time?: number\n end_time?: number\n confidence?: number\n}\n\n/** One recognised utterance. `start_time` / `end_time` are milliseconds. */\nexport interface BytePlusASRUtterance {\n text?: string\n start_time?: number\n end_time?: number\n words?: Array<BytePlusASRWord>\n /**\n * Extra per-utterance annotations. Speaker labels arrive here when\n * `enable_speaker_info` is set; the exact key is read defensively because it\n * could not be confirmed against a live response.\n */\n additions?: Record<string, string>\n}\n\nexport interface BytePlusASRResult {\n text?: string\n utterances?: Array<BytePlusASRUtterance>\n}\n\n/**\n * Response body for `POST /api/v3/auc/bigmodel/recognize/flash`.\n *\n * The Volcengine-lineage wire shape nests everything under `result`; BytePlus'\n * prose docs describe the same payload as \"transcript + utterances\", so the\n * flat spelling is tolerated as a fallback.\n */\nexport interface BytePlusASRRecognizeResponse {\n /** `duration` is the audio length in **milliseconds**. */\n audio_info?: { duration?: number }\n result?: BytePlusASRResult\n /** Flat alias for `result.text`. */\n transcript?: string\n /** Flat alias for `result.utterances`. */\n utterances?: Array<BytePlusASRUtterance>\n}\n\n// ============================================================================\n// Errors\n// ============================================================================\n\n/**\n * Seed Speech error envelope: a flat numeric `code` plus a `message`, e.g.\n * `{\"code\": 45000010, \"message\": \"Invalid X-Api-Key\"}` (verified live on a\n * 401). Format it with `bytePlusVoiceError` from `../utils/client`.\n */\nexport interface BytePlusVoiceErrorBody {\n code?: number\n message?: string\n}\n"],"mappings":";;;;;;;;AAyCA,IAAa,4BAA4B;CACvC;CAAM;CAAO;CAAO;CAAO;CAAO;AACpC;;;;;;AAmJA,IAAa,2BAA2B;;AAGxC,IAAa,+BAA+B"}
@@ -0,0 +1,165 @@
1
+ import { BytePlusImageOutputFormat, BytePlusImageResponseFormat, BytePlusOptimizePromptOptions, BytePlusSequentialImageGeneration, BytePlusSequentialImageGenerationOptions } from './wire-types.js';
2
+ import { BytePlusImageModel, BytePlusImageSize } from '../model-meta.js';
3
+ /**
4
+ * BytePlus documents a 600-word ceiling on the image prompt. Word-based, so
5
+ * it is only meaningful for space-separated scripts — the check below never
6
+ * fires for Chinese or Japanese text, which is the intended behaviour.
7
+ */
8
+ export declare const BYTEPLUS_IMAGE_MAX_PROMPT_WORDS = 600;
9
+ /**
10
+ * Upper bound of `sequential_image_generation_options.max_images`, i.e. the
11
+ * most images one request can return.
12
+ */
13
+ export declare const BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES = 15;
14
+ /**
15
+ * Models that accept `output_format`.
16
+ *
17
+ * The Ark OpenAPI document's field note claims 5.0-lite only, but its own
18
+ * request demo sends `output_format` on `seedream-5-0-260128`, so the whole
19
+ * 5.0 family is treated as supporting it. The Seedream 4.x snapshots are
20
+ * documented as not reading it, so `output_format` is omitted from their
21
+ * provider-options type — but, as with `sequential_image_generation`, it is
22
+ * not gated at runtime: a value that reaches a model which does not read it
23
+ * comes back as an Ark error rather than a local rejection.
24
+ */
25
+ export declare const BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS: ReadonlyArray<BytePlusImageModel>;
26
+ /**
27
+ * Base provider options shared by every Seedream model.
28
+ */
29
+ export interface BytePlusImageBaseProviderOptions {
30
+ /**
31
+ * Return images as expiring links (`url`, valid 24 hours) or inline base64
32
+ * (`b64_json`).
33
+ *
34
+ * @default 'url'
35
+ */
36
+ response_format?: BytePlusImageResponseFormat;
37
+ /**
38
+ * Whether to stamp an "AI generated" watermark in the bottom-right corner.
39
+ *
40
+ * **BytePlus defaults this to `true`.** Pass `false` for a clean image.
41
+ */
42
+ watermark?: boolean;
43
+ /**
44
+ * Group-image mode. Set to `auto` to let the model return a set of related
45
+ * images (bounded by {@link BytePlusImageBaseProviderOptions.sequential_image_generation_options}).
46
+ * `generateImage()`'s `numberOfImages` sets this for you; an explicit value
47
+ * here wins.
48
+ *
49
+ * Documented on Seedream 5.0-lite, 4.5 and 4.0. It is sent as given on
50
+ * every model rather than gated locally — the shipped 5.0 ids post-date the
51
+ * published parameter table, and an unsupported combination comes back as a
52
+ * clear Ark error.
53
+ *
54
+ * @default 'disabled'
55
+ */
56
+ sequential_image_generation?: BytePlusSequentialImageGeneration;
57
+ /** Bounds for group-image mode. Only read when the mode is `auto`. */
58
+ sequential_image_generation_options?: BytePlusSequentialImageGenerationOptions;
59
+ /**
60
+ * Prompt-rewriting configuration. Documented on Seedream 5.0-lite, 4.5 and
61
+ * 4.0; `mode: 'fast'` is unsupported on 5.0-lite and 4.5.
62
+ */
63
+ optimize_prompt_options?: BytePlusOptimizePromptOptions;
64
+ }
65
+ /**
66
+ * Provider options for the Seedream 5.0 family, which additionally chooses the
67
+ * generated file format.
68
+ */
69
+ export interface BytePlusSeedream5ImageProviderOptions extends BytePlusImageBaseProviderOptions {
70
+ /**
71
+ * File format of the generated image.
72
+ *
73
+ * @default 'jpeg'
74
+ */
75
+ output_format?: BytePlusImageOutputFormat;
76
+ }
77
+ /**
78
+ * Every Seedream provider option, used as the adapter's base option type.
79
+ * Call sites are narrowed per model by
80
+ * {@link BytePlusImageModelProviderOptionsByName}.
81
+ */
82
+ export type BytePlusImageProviderOptions = BytePlusSeedream5ImageProviderOptions;
83
+ /**
84
+ * Type-only map from image model name to its provider options.
85
+ */
86
+ export type BytePlusImageModelProviderOptionsByName = {
87
+ 'dola-seedream-5-0-pro-260628': BytePlusSeedream5ImageProviderOptions;
88
+ 'seedream-5-0-260128': BytePlusSeedream5ImageProviderOptions;
89
+ 'seedream-5-0-lite-260128': BytePlusSeedream5ImageProviderOptions;
90
+ 'seedream-4-5-251128': BytePlusImageBaseProviderOptions;
91
+ 'seedream-4-0-250828': BytePlusImageBaseProviderOptions;
92
+ };
93
+ /**
94
+ * Type-only map from image model name to the non-text prompt modalities it
95
+ * accepts. Every shipped Seedream model takes reference images for
96
+ * image-conditioned generation.
97
+ */
98
+ export type BytePlusImageModelInputModalitiesByName = {
99
+ [K in BytePlusImageModel]: readonly ['image'];
100
+ };
101
+ /**
102
+ * A parsed `size` value: either the shorthand token form or explicit pixels.
103
+ */
104
+ export type ParsedBytePlusImageSize = {
105
+ kind: 'token';
106
+ value: '1K' | '2K' | '4K';
107
+ } | {
108
+ kind: 'pixels';
109
+ width: number;
110
+ height: number;
111
+ };
112
+ /**
113
+ * Parses a Seedream `size` string. Accepts a shorthand token (case-insensitive
114
+ * — `2k` normalizes to `2K`) or explicit `WIDTHxHEIGHT` pixels, and returns
115
+ * `undefined` for anything else, including mixtures such as `2K x 1024`.
116
+ */
117
+ export declare function parseBytePlusImageSize(size: string): ParsedBytePlusImageSize | undefined;
118
+ /**
119
+ * Validates the generic `size` option and returns the string to put on the
120
+ * wire (`2K`, `2048x2048`), or `undefined` when no size was requested.
121
+ *
122
+ * This checks the *form* only. Which pixel dimensions a given model actually
123
+ * accepts is not encoded here, so the message deliberately makes no per-model
124
+ * claim; an out-of-range size is left to the API to reject.
125
+ *
126
+ * @throws Error when the value is neither a size token nor `WIDTHxHEIGHT`.
127
+ */
128
+ export declare function resolveBytePlusImageSize(size: BytePlusImageSize | string | undefined): string | undefined;
129
+ /**
130
+ * Validates the prompt text against BytePlus's documented limits.
131
+ *
132
+ * @throws Error when the prompt is empty or exceeds
133
+ * {@link BYTEPLUS_IMAGE_MAX_PROMPT_WORDS} words.
134
+ */
135
+ export declare function validateBytePlusImagePrompt(model: string, prompt: string): void;
136
+ /**
137
+ * Validates the reference-image count against the model's editing limit.
138
+ *
139
+ * A model this package has no limit for is left to Ark, deliberately and
140
+ * explicitly. `model` is typed closed, but `wire-types.ts` documents that the
141
+ * endpoint also accepts preconfigured endpoint ids (`ep-…`), so a JS caller
142
+ * can reach an id that is not in the table. Reading `undefined` out of it and
143
+ * comparing `count > undefined` — always false — would disable the guard by
144
+ * accident and look identical to passing; the explicit early return says the
145
+ * skip is intended.
146
+ *
147
+ * @throws Error when more references are supplied than a *known* model accepts.
148
+ */
149
+ export declare function validateBytePlusReferenceImages(model: BytePlusImageModel, count: number): void;
150
+ /**
151
+ * Maps the generic `numberOfImages` option onto Seedream's group-image
152
+ * parameters.
153
+ *
154
+ * The endpoint has no `n`: more than one image per request is only reachable
155
+ * through `sequential_image_generation: 'auto'`, where `max_images` is an
156
+ * upper bound and the model decides how many images the prompt actually
157
+ * warrants. A request for N images can therefore come back with fewer — the
158
+ * one place BytePlus cannot honour `numberOfImages` exactly.
159
+ *
160
+ * @throws Error when the count is not an integer in `[1, 15]`.
161
+ */
162
+ export declare function resolveBytePlusSequentialImages(model: string, numberOfImages: number | undefined): {
163
+ sequential_image_generation?: BytePlusSequentialImageGeneration;
164
+ sequential_image_generation_options?: BytePlusSequentialImageGenerationOptions;
165
+ };