@tanstack/ai-byteplus 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +202 -0
- package/dist/esm/adapters/image.d.ts +89 -0
- package/dist/esm/adapters/image.js +229 -0
- package/dist/esm/adapters/image.js.map +1 -0
- package/dist/esm/adapters/text.d.ts +163 -0
- package/dist/esm/adapters/text.js +347 -0
- package/dist/esm/adapters/text.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +102 -0
- package/dist/esm/adapters/transcription.js +274 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +143 -0
- package/dist/esm/adapters/tts.js +307 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/adapters/video.d.ts +182 -0
- package/dist/esm/adapters/video.js +442 -0
- package/dist/esm/adapters/video.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
- package/dist/esm/audio/tts-provider-options.d.ts +114 -0
- package/dist/esm/audio/wire-types.d.ts +261 -0
- package/dist/esm/audio/wire-types.js +28 -0
- package/dist/esm/audio/wire-types.js.map +1 -0
- package/dist/esm/image/image-provider-options.d.ts +165 -0
- package/dist/esm/image/image-provider-options.js +134 -0
- package/dist/esm/image/image-provider-options.js.map +1 -0
- package/dist/esm/image/wire-types.d.ts +149 -0
- package/dist/esm/index.d.ts +25 -0
- package/dist/esm/index.js +11 -0
- package/dist/esm/message-types.d.ts +154 -0
- package/dist/esm/model-meta.d.ts +594 -0
- package/dist/esm/model-meta.js +619 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/text/text-provider-options.d.ts +109 -0
- package/dist/esm/utils/client.d.ts +183 -0
- package/dist/esm/utils/client.js +253 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +197 -0
- package/dist/esm/video/video-provider-options.js +191 -0
- package/dist/esm/video/video-provider-options.js.map +1 -0
- package/dist/esm/video/wire-types.d.ts +248 -0
- package/package.json +77 -0
- package/src/adapters/image.ts +409 -0
- package/src/adapters/text.ts +539 -0
- package/src/adapters/transcription.ts +479 -0
- package/src/adapters/tts.ts +447 -0
- package/src/adapters/video.ts +732 -0
- package/src/audio/transcription-provider-options.ts +46 -0
- package/src/audio/tts-provider-options.ts +122 -0
- package/src/audio/wire-types.ts +290 -0
- package/src/image/image-provider-options.ts +288 -0
- package/src/image/wire-types.ts +169 -0
- package/src/index.ts +222 -0
- package/src/message-types.ts +169 -0
- package/src/model-meta.ts +954 -0
- package/src/text/text-provider-options.ts +151 -0
- package/src/utils/client.ts +377 -0
- package/src/video/video-provider-options.ts +361 -0
- package/src/video/wire-types.ts +293 -0
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import { BytePlusTTSAudioFormat, BytePlusTTSReference, BytePlusTTSSampleRate, BytePlusTTSSubtitle } from './wire-types.js';
|
|
2
|
+
import { TTSResult } from '@tanstack/ai';
|
|
3
|
+
/**
|
|
4
|
+
* Seed Speech voice identifier (`speaker` on the wire).
|
|
5
|
+
*
|
|
6
|
+
* Voice ids encode language, gender, character name and model generation:
|
|
7
|
+
* `en_female_stokie_uranus_bigtts` is the English female "Stokie" voice on
|
|
8
|
+
* TTS 2.0. The generation suffix matters when picking one:
|
|
9
|
+
*
|
|
10
|
+
* - `_uranus_bigtts` — TTS 2.0 voices (the current generation).
|
|
11
|
+
* - `_mars_bigtts` / `_moon_bigtts` — TTS 1.0 voices.
|
|
12
|
+
* - `*_emo_v2_*` — TTS 1.0 voices that additionally accept emotion tags.
|
|
13
|
+
*
|
|
14
|
+
* The full roster lives at
|
|
15
|
+
* https://docs.byteplus.com/en/docs/byteplusvoice/voicelist and changes far
|
|
16
|
+
* more often than this package ships, so the union stays open: any string is
|
|
17
|
+
* accepted, and the one listed id is the adapter's default.
|
|
18
|
+
*/
|
|
19
|
+
export type BytePlusTTSVoice = 'en_female_stokie_uranus_bigtts' | (string & {});
|
|
20
|
+
/**
|
|
21
|
+
* Provider-specific options for BytePlus Seed Speech TTS
|
|
22
|
+
* (`POST /api/v3/tts/create`).
|
|
23
|
+
*
|
|
24
|
+
* These map 1:1 onto the wire fields so the BytePlus documentation stays
|
|
25
|
+
* useful; where a cross-provider `TTSOptions` field covers the same ground
|
|
26
|
+
* (`voice`, `format`, `speed`), the option here wins.
|
|
27
|
+
*/
|
|
28
|
+
export interface BytePlusTTSProviderOptions {
|
|
29
|
+
/**
|
|
30
|
+
* Voice id. Overrides `TTSOptions.voice` when both are set. It is sent as
|
|
31
|
+
* the `speaker` of a single `references` entry.
|
|
32
|
+
*/
|
|
33
|
+
speaker?: BytePlusTTSVoice;
|
|
34
|
+
/**
|
|
35
|
+
* Full `references` array, for voice cloning or image-referenced delivery.
|
|
36
|
+
* When set this replaces the entry the adapter would otherwise build from
|
|
37
|
+
* `speaker` / `TTSOptions.voice`, so include a `speaker` entry yourself if
|
|
38
|
+
* you still want a stock voice. Address audio references from the text with
|
|
39
|
+
* `@Audio1`..`@Audio3`.
|
|
40
|
+
*/
|
|
41
|
+
references?: Array<BytePlusTTSReference>;
|
|
42
|
+
/**
|
|
43
|
+
* Output format. Overrides the mapping applied to `TTSOptions.format`,
|
|
44
|
+
* which is useful for `ogg_opus` and for pinning `pcm` explicitly.
|
|
45
|
+
*/
|
|
46
|
+
format?: BytePlusTTSAudioFormat;
|
|
47
|
+
/**
|
|
48
|
+
* Output sample rate in Hz. Defaults to 24000 — the adapter always sends an
|
|
49
|
+
* explicit rate because the documented server default (40000) is not one of
|
|
50
|
+
* the values the endpoint accepts. For `pcm` output this is also what the
|
|
51
|
+
* returned `contentType` (`audio/L16;rate=…`) reports.
|
|
52
|
+
*/
|
|
53
|
+
sample_rate?: BytePlusTTSSampleRate;
|
|
54
|
+
/**
|
|
55
|
+
* Pitch adjustment in the range `-12`..`12`, where `0` is the voice's
|
|
56
|
+
* natural pitch.
|
|
57
|
+
*/
|
|
58
|
+
pitch_rate?: number;
|
|
59
|
+
/**
|
|
60
|
+
* Speaking rate in the range `-50`..`100` (`-50` = 0.5×, `0` = 1×,
|
|
61
|
+
* `100` = 2×). Overrides the value derived from `TTSOptions.speed`.
|
|
62
|
+
*/
|
|
63
|
+
speech_rate?: number;
|
|
64
|
+
/**
|
|
65
|
+
* Loudness adjustment in the range `-50`..`100`, where `0` is the voice's
|
|
66
|
+
* natural level.
|
|
67
|
+
*/
|
|
68
|
+
loudness_rate?: number;
|
|
69
|
+
/**
|
|
70
|
+
* Ask for sentence- and word-level timings alongside the audio. They are
|
|
71
|
+
* surfaced on {@link BytePlusTTSResult.subtitle}.
|
|
72
|
+
*/
|
|
73
|
+
enable_subtitle?: boolean;
|
|
74
|
+
/**
|
|
75
|
+
* Watermark the generated audio. The field name is confirmed against the
|
|
76
|
+
* endpoint schema; the boolean type is assumed and unprobed.
|
|
77
|
+
*/
|
|
78
|
+
watermark?: boolean;
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* BytePlus-specific extension of `TTSResult`.
|
|
82
|
+
*
|
|
83
|
+
* The cross-provider `TTSResult` has nowhere to put the subtitle timings or
|
|
84
|
+
* the temporary download URL, so callers who want them narrow the result:
|
|
85
|
+
*
|
|
86
|
+
* ```ts
|
|
87
|
+
* const result: BytePlusTTSResult = await generateSpeech({ adapter, text })
|
|
88
|
+
* for (const sentence of result.subtitle?.sentences ?? []) {
|
|
89
|
+
* console.log(sentence.text, sentence.start_time)
|
|
90
|
+
* }
|
|
91
|
+
* ```
|
|
92
|
+
*/
|
|
93
|
+
export interface BytePlusTTSResult extends TTSResult {
|
|
94
|
+
/**
|
|
95
|
+
* Sentence and word timings, present only when
|
|
96
|
+
* `modelOptions.enable_subtitle` was set. Their `start_time` / `end_time`
|
|
97
|
+
* are **milliseconds**, even though `duration` and
|
|
98
|
+
* {@link BytePlusTTSResult.originalDuration} are seconds.
|
|
99
|
+
*/
|
|
100
|
+
subtitle?: BytePlusTTSSubtitle;
|
|
101
|
+
/**
|
|
102
|
+
* Length of the audio in seconds *before* `speech_rate` was applied. This
|
|
103
|
+
* is what BytePlus bills on and what the 120 s cap applies to, so it is the
|
|
104
|
+
* number to meter against — `duration` reflects the delivered clip and can
|
|
105
|
+
* exceed 120 s when the speech is slowed down.
|
|
106
|
+
*/
|
|
107
|
+
originalDuration?: number;
|
|
108
|
+
/**
|
|
109
|
+
* Temporary download URL for the same audio that `audio` carries as base64.
|
|
110
|
+
* **Expires roughly 2 hours after generation** — persist the bytes, not the
|
|
111
|
+
* link.
|
|
112
|
+
*/
|
|
113
|
+
url?: string;
|
|
114
|
+
}
|
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Minimal wire types for the BytePlus **Seed Speech** HTTP API (TTS + ASR).
|
|
3
|
+
*
|
|
4
|
+
* Seed Speech is a separate product from Ark: it lives on
|
|
5
|
+
* `voice.ap-southeast-1.bytepluses.com`, authenticates with `X-Api-Key`
|
|
6
|
+
* (a different key from `ARK_API_KEY`), and returns a flat numeric error
|
|
7
|
+
* envelope instead of Ark's OpenAI-shaped one.
|
|
8
|
+
*
|
|
9
|
+
* Only the fields the adapters read or write are modelled here — this is a
|
|
10
|
+
* hand-written subset, not a generated schema.
|
|
11
|
+
*
|
|
12
|
+
* Provenance:
|
|
13
|
+
* - Endpoints, auth header, format/rate ranges and the 120 s TTS output cap:
|
|
14
|
+
* BytePlus Seed Speech docs (`docs.byteplus.com/en/docs/byteplusvoice`),
|
|
15
|
+
* captured in the Phase 0 research notes.
|
|
16
|
+
* - Error envelope `{code, message}`: verified live — an Ark key sent as
|
|
17
|
+
* `X-Api-Key` returns HTTP 401 `{"code":45000010,"message":"Invalid X-Api-Key"}`.
|
|
18
|
+
* - ASR request/response shape (`user`/`audio`/`request` in, `audio_info` +
|
|
19
|
+
* `result.utterances` out, all timings in **milliseconds**): the Volcengine
|
|
20
|
+
* flash-recognition reference the BytePlus endpoint is derived from
|
|
21
|
+
* (`docs.volcengine.com/docs/6561/1631584`).
|
|
22
|
+
*
|
|
23
|
+
* No Seed Speech API key was available when these were written, so the TTS
|
|
24
|
+
* response fields are documented-but-unverified; the adapters parse them
|
|
25
|
+
* defensively rather than assuming they are always present.
|
|
26
|
+
*/
|
|
27
|
+
/** Output container/codec accepted by `audio_config.format`. */
|
|
28
|
+
export type BytePlusTTSAudioFormat = 'wav' | 'mp3' | 'pcm' | 'ogg_opus';
|
|
29
|
+
/**
|
|
30
|
+
* Sample rates `audio_config.sample_rate` accepts.
|
|
31
|
+
*
|
|
32
|
+
* The docs also state a *default* of 40000, which is not one of the valid
|
|
33
|
+
* values — a documentation bug. The adapter therefore always sends an
|
|
34
|
+
* explicit rate rather than relying on the server default.
|
|
35
|
+
*/
|
|
36
|
+
export declare const BYTEPLUS_TTS_SAMPLE_RATES: readonly [8000, 16000, 24000, 32000, 44100, 48000];
|
|
37
|
+
/** A sample rate `audio_config.sample_rate` accepts. */
|
|
38
|
+
export type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number];
|
|
39
|
+
/**
|
|
40
|
+
* One entry of the request's `references` array.
|
|
41
|
+
*
|
|
42
|
+
* This is where the voice lives: `speaker` names a stock voice, while
|
|
43
|
+
* `audio_url` / `audio_data` supply reference clips to clone (the `@Audio1`..
|
|
44
|
+
* `@Audio3` markers in `text_prompt` address them positionally). Exactly one
|
|
45
|
+
* of `speaker`, `audio_url` or `audio_data` may be set per entry; at most 3
|
|
46
|
+
* audio references (30 s / 10 MB each) and 1 image reference (10 MB), and
|
|
47
|
+
* image references are mutually exclusive with audio ones.
|
|
48
|
+
*
|
|
49
|
+
* **Member object shape is unresolved — must live-probe when the Seed Speech
|
|
50
|
+
* key lands.** The docs list the member fields flat (`speaker | audio_data |
|
|
51
|
+
* audio_url | image_data | image_url`) without a worked example, so whether
|
|
52
|
+
* the server wants a flat `{ speaker }` or a discriminated `{ type, ... }`
|
|
53
|
+
* could not be settled. The adapter sends the flat reading — see
|
|
54
|
+
* `buildTTSRequestBody` in `../adapters/tts`.
|
|
55
|
+
*/
|
|
56
|
+
export interface BytePlusTTSReference {
|
|
57
|
+
/** Stock voice id, e.g. `en_female_stokie_uranus_bigtts`. */
|
|
58
|
+
speaker?: string;
|
|
59
|
+
/** URL of a reference clip to clone (≤30 s, ≤10 MB). */
|
|
60
|
+
audio_url?: string;
|
|
61
|
+
/** Base64 reference clip to clone (≤30 s, ≤10 MB). */
|
|
62
|
+
audio_data?: string;
|
|
63
|
+
/** URL of a reference image (≤10 MB). Mutually exclusive with audio refs. */
|
|
64
|
+
image_url?: string;
|
|
65
|
+
/** Base64 reference image (≤10 MB). Mutually exclusive with audio refs. */
|
|
66
|
+
image_data?: string;
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* `audio_config` block of a TTS request — exactly six fields.
|
|
70
|
+
*
|
|
71
|
+
* The three `*_rate` fields are integer percentages relative to the voice's
|
|
72
|
+
* neutral delivery, not multipliers.
|
|
73
|
+
*/
|
|
74
|
+
export interface BytePlusTTSAudioConfig {
|
|
75
|
+
/** Output format. Defaults to `wav` server-side. */
|
|
76
|
+
format?: BytePlusTTSAudioFormat;
|
|
77
|
+
/**
|
|
78
|
+
* Output sample rate in Hz. See {@link BYTEPLUS_TTS_SAMPLE_RATES} for the
|
|
79
|
+
* valid values and for why the adapter always sends one.
|
|
80
|
+
*/
|
|
81
|
+
sample_rate?: number;
|
|
82
|
+
/** Speaking rate, `-50`..`100`. `-50` = 0.5×, `0` = 1×, `100` = 2×. */
|
|
83
|
+
speech_rate?: number;
|
|
84
|
+
/** Loudness adjustment, `-50`..`100`. `0` is the voice's natural level. */
|
|
85
|
+
loudness_rate?: number;
|
|
86
|
+
/** Pitch adjustment, `-12`..`12`. `0` is the voice's natural pitch. */
|
|
87
|
+
pitch_rate?: number;
|
|
88
|
+
/** Emit sentence and word timings in the response. Defaults to `false`. */
|
|
89
|
+
enable_subtitle?: boolean;
|
|
90
|
+
}
|
|
91
|
+
/** Request body for `POST /api/v3/tts/create` — exactly five fields. */
|
|
92
|
+
export interface BytePlusTTSCreateRequest {
|
|
93
|
+
/** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */
|
|
94
|
+
model: string;
|
|
95
|
+
/**
|
|
96
|
+
* The text to speak (≤3000 chars). Dual-purpose: either literal text or a
|
|
97
|
+
* natural-language description of the delivery, and the place the
|
|
98
|
+
* `@Audio1`..`@Audio3` reference markers go.
|
|
99
|
+
*
|
|
100
|
+
* **`text_prompt` is correct — do not "fix" this to `text`.** This endpoint
|
|
101
|
+
* has no request-side `text` field. The `text` spelling belongs to the
|
|
102
|
+
* *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits under
|
|
103
|
+
* `req_params.text`. Confirmed against
|
|
104
|
+
* docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01, which lists this
|
|
105
|
+
* body as exactly `model`, `text_prompt`, `references`, `audio_config`,
|
|
106
|
+
* `watermark`.
|
|
107
|
+
*/
|
|
108
|
+
text_prompt: string;
|
|
109
|
+
/** Voice selection and cloning references. See {@link BytePlusTTSReference}. */
|
|
110
|
+
references?: Array<BytePlusTTSReference>;
|
|
111
|
+
audio_config?: BytePlusTTSAudioConfig;
|
|
112
|
+
/**
|
|
113
|
+
* Watermark the generated audio. The field name is confirmed; the boolean
|
|
114
|
+
* type is assumed by analogy with Seedream's `watermark` and unprobed.
|
|
115
|
+
*/
|
|
116
|
+
watermark?: boolean;
|
|
117
|
+
}
|
|
118
|
+
/**
|
|
119
|
+
* One timed entry of a TTS subtitle track.
|
|
120
|
+
*
|
|
121
|
+
* **Times are milliseconds** — unlike the response's `duration` fields, which
|
|
122
|
+
* are seconds. The endpoint genuinely mixes units.
|
|
123
|
+
*/
|
|
124
|
+
export interface BytePlusTTSSubtitleEntry {
|
|
125
|
+
text?: string;
|
|
126
|
+
start_time?: number;
|
|
127
|
+
end_time?: number;
|
|
128
|
+
}
|
|
129
|
+
/** Sentence- and word-level timings returned when `enable_subtitle` is set. */
|
|
130
|
+
export interface BytePlusTTSSubtitle {
|
|
131
|
+
sentences?: Array<BytePlusTTSSubtitleEntry>;
|
|
132
|
+
words?: Array<BytePlusTTSSubtitleEntry>;
|
|
133
|
+
}
|
|
134
|
+
/** Response body for `POST /api/v3/tts/create`. */
|
|
135
|
+
export interface BytePlusTTSCreateResponse {
|
|
136
|
+
/**
|
|
137
|
+
* Status code — `0` on success, a flat error code otherwise.
|
|
138
|
+
*
|
|
139
|
+
* Typed as `number | string` because only the *error* envelope was verified
|
|
140
|
+
* live (HTTP 401 `{"code":45000010,…}`); the success envelope's shape is
|
|
141
|
+
* docs-derived and no voice key was available to confirm it. Treating a
|
|
142
|
+
* string code as "not a failure" would return a failed 200 as success, so
|
|
143
|
+
* the adapter coerces before comparing — see `isZeroCode` in
|
|
144
|
+
* `adapters/tts.ts`.
|
|
145
|
+
*/
|
|
146
|
+
code?: number | string;
|
|
147
|
+
message?: string;
|
|
148
|
+
/** Base64-encoded audio in the requested `audio_config.format`. */
|
|
149
|
+
audio?: string;
|
|
150
|
+
/**
|
|
151
|
+
* Length of the delivered audio in **seconds** (float), after `speech_rate`
|
|
152
|
+
* is applied. This can legitimately exceed 120 when the clip is slowed
|
|
153
|
+
* down — the 120 s cap applies to {@link BytePlusTTSCreateResponse.original_duration}.
|
|
154
|
+
*/
|
|
155
|
+
duration?: number | string;
|
|
156
|
+
/**
|
|
157
|
+
* Length in **seconds** (float) before rate adjustment. This is the billing
|
|
158
|
+
* basis and is capped at 120.
|
|
159
|
+
*/
|
|
160
|
+
original_duration?: number | string;
|
|
161
|
+
/** Temporary download URL for the same audio. Expires after ~2 hours. */
|
|
162
|
+
url?: string;
|
|
163
|
+
/** Sentence and word timings, present when `enable_subtitle` was set. */
|
|
164
|
+
subtitle?: BytePlusTTSSubtitle;
|
|
165
|
+
}
|
|
166
|
+
/**
|
|
167
|
+
* Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo
|
|
168
|
+
* model. The flash endpoint takes no `model` field in its body — the model is
|
|
169
|
+
* chosen entirely by this header.
|
|
170
|
+
*/
|
|
171
|
+
export declare const BYTEPLUS_ASR_RESOURCE_ID = "volc.seedasr.auc_turbo";
|
|
172
|
+
/** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */
|
|
173
|
+
export declare const BYTEPLUS_ASR_RESOURCE_HEADER = "X-Api-Resource-Id";
|
|
174
|
+
/**
|
|
175
|
+
* Audio input. Exactly one of `url` or `data` is sent — the endpoint accepts
|
|
176
|
+
* files up to 2 hours long / 100 MB.
|
|
177
|
+
*/
|
|
178
|
+
export interface BytePlusASRAudio {
|
|
179
|
+
/** Publicly reachable URL of the audio file. */
|
|
180
|
+
url?: string;
|
|
181
|
+
/** Base64-encoded audio bytes. */
|
|
182
|
+
data?: string;
|
|
183
|
+
/** Container hint, e.g. `mp3`, `wav`, `ogg`. */
|
|
184
|
+
format?: string;
|
|
185
|
+
}
|
|
186
|
+
/** `request` block of a recognition call. */
|
|
187
|
+
export interface BytePlusASRRequestOptions {
|
|
188
|
+
/** Recognition model family. Defaults to `bigmodel`. */
|
|
189
|
+
model_name?: string;
|
|
190
|
+
/** Inverse text normalisation (spoken numbers → digits). */
|
|
191
|
+
enable_itn?: boolean;
|
|
192
|
+
/** Insert punctuation. */
|
|
193
|
+
enable_punc?: boolean;
|
|
194
|
+
/** Disfluency removal ("um", repeated words). */
|
|
195
|
+
enable_ddc?: boolean;
|
|
196
|
+
/** Attach per-utterance speaker labels. */
|
|
197
|
+
enable_speaker_info?: boolean;
|
|
198
|
+
/** Return the `utterances` breakdown as well as the flat transcript. */
|
|
199
|
+
show_utterances?: boolean;
|
|
200
|
+
/** Spoken language hint, e.g. `en-US`. */
|
|
201
|
+
language?: string;
|
|
202
|
+
}
|
|
203
|
+
/** Request body for `POST /api/v3/auc/bigmodel/recognize/flash`. */
|
|
204
|
+
export interface BytePlusASRRecognizeRequest {
|
|
205
|
+
user?: {
|
|
206
|
+
uid?: string;
|
|
207
|
+
};
|
|
208
|
+
audio: BytePlusASRAudio;
|
|
209
|
+
request?: BytePlusASRRequestOptions;
|
|
210
|
+
}
|
|
211
|
+
/** One recognised word. `start_time` / `end_time` are milliseconds. */
|
|
212
|
+
export interface BytePlusASRWord {
|
|
213
|
+
text?: string;
|
|
214
|
+
start_time?: number;
|
|
215
|
+
end_time?: number;
|
|
216
|
+
confidence?: number;
|
|
217
|
+
}
|
|
218
|
+
/** One recognised utterance. `start_time` / `end_time` are milliseconds. */
|
|
219
|
+
export interface BytePlusASRUtterance {
|
|
220
|
+
text?: string;
|
|
221
|
+
start_time?: number;
|
|
222
|
+
end_time?: number;
|
|
223
|
+
words?: Array<BytePlusASRWord>;
|
|
224
|
+
/**
|
|
225
|
+
* Extra per-utterance annotations. Speaker labels arrive here when
|
|
226
|
+
* `enable_speaker_info` is set; the exact key is read defensively because it
|
|
227
|
+
* could not be confirmed against a live response.
|
|
228
|
+
*/
|
|
229
|
+
additions?: Record<string, string>;
|
|
230
|
+
}
|
|
231
|
+
export interface BytePlusASRResult {
|
|
232
|
+
text?: string;
|
|
233
|
+
utterances?: Array<BytePlusASRUtterance>;
|
|
234
|
+
}
|
|
235
|
+
/**
|
|
236
|
+
* Response body for `POST /api/v3/auc/bigmodel/recognize/flash`.
|
|
237
|
+
*
|
|
238
|
+
* The Volcengine-lineage wire shape nests everything under `result`; BytePlus'
|
|
239
|
+
* prose docs describe the same payload as "transcript + utterances", so the
|
|
240
|
+
* flat spelling is tolerated as a fallback.
|
|
241
|
+
*/
|
|
242
|
+
export interface BytePlusASRRecognizeResponse {
|
|
243
|
+
/** `duration` is the audio length in **milliseconds**. */
|
|
244
|
+
audio_info?: {
|
|
245
|
+
duration?: number;
|
|
246
|
+
};
|
|
247
|
+
result?: BytePlusASRResult;
|
|
248
|
+
/** Flat alias for `result.text`. */
|
|
249
|
+
transcript?: string;
|
|
250
|
+
/** Flat alias for `result.utterances`. */
|
|
251
|
+
utterances?: Array<BytePlusASRUtterance>;
|
|
252
|
+
}
|
|
253
|
+
/**
|
|
254
|
+
* Seed Speech error envelope: a flat numeric `code` plus a `message`, e.g.
|
|
255
|
+
* `{"code": 45000010, "message": "Invalid X-Api-Key"}` (verified live on a
|
|
256
|
+
* 401). Format it with `bytePlusVoiceError` from `../utils/client`.
|
|
257
|
+
*/
|
|
258
|
+
export interface BytePlusVoiceErrorBody {
|
|
259
|
+
code?: number;
|
|
260
|
+
message?: string;
|
|
261
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
//#region src/audio/wire-types.ts
|
|
2
|
+
/**
|
|
3
|
+
* Sample rates `audio_config.sample_rate` accepts.
|
|
4
|
+
*
|
|
5
|
+
* The docs also state a *default* of 40000, which is not one of the valid
|
|
6
|
+
* values — a documentation bug. The adapter therefore always sends an
|
|
7
|
+
* explicit rate rather than relying on the server default.
|
|
8
|
+
*/
|
|
9
|
+
var BYTEPLUS_TTS_SAMPLE_RATES = [
|
|
10
|
+
8e3,
|
|
11
|
+
16e3,
|
|
12
|
+
24e3,
|
|
13
|
+
32e3,
|
|
14
|
+
44100,
|
|
15
|
+
48e3
|
|
16
|
+
];
|
|
17
|
+
/**
|
|
18
|
+
* Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo
|
|
19
|
+
* model. The flash endpoint takes no `model` field in its body — the model is
|
|
20
|
+
* chosen entirely by this header.
|
|
21
|
+
*/
|
|
22
|
+
var BYTEPLUS_ASR_RESOURCE_ID = "volc.seedasr.auc_turbo";
|
|
23
|
+
/** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */
|
|
24
|
+
var BYTEPLUS_ASR_RESOURCE_HEADER = "X-Api-Resource-Id";
|
|
25
|
+
//#endregion
|
|
26
|
+
export { BYTEPLUS_ASR_RESOURCE_HEADER, BYTEPLUS_ASR_RESOURCE_ID, BYTEPLUS_TTS_SAMPLE_RATES };
|
|
27
|
+
|
|
28
|
+
//# sourceMappingURL=wire-types.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"wire-types.js","names":[],"sources":["../../../src/audio/wire-types.ts"],"sourcesContent":["/**\n * Minimal wire types for the BytePlus **Seed Speech** HTTP API (TTS + ASR).\n *\n * Seed Speech is a separate product from Ark: it lives on\n * `voice.ap-southeast-1.bytepluses.com`, authenticates with `X-Api-Key`\n * (a different key from `ARK_API_KEY`), and returns a flat numeric error\n * envelope instead of Ark's OpenAI-shaped one.\n *\n * Only the fields the adapters read or write are modelled here — this is a\n * hand-written subset, not a generated schema.\n *\n * Provenance:\n * - Endpoints, auth header, format/rate ranges and the 120 s TTS output cap:\n * BytePlus Seed Speech docs (`docs.byteplus.com/en/docs/byteplusvoice`),\n * captured in the Phase 0 research notes.\n * - Error envelope `{code, message}`: verified live — an Ark key sent as\n * `X-Api-Key` returns HTTP 401 `{\"code\":45000010,\"message\":\"Invalid X-Api-Key\"}`.\n * - ASR request/response shape (`user`/`audio`/`request` in, `audio_info` +\n * `result.utterances` out, all timings in **milliseconds**): the Volcengine\n * flash-recognition reference the BytePlus endpoint is derived from\n * (`docs.volcengine.com/docs/6561/1631584`).\n *\n * No Seed Speech API key was available when these were written, so the TTS\n * response fields are documented-but-unverified; the adapters parse them\n * defensively rather than assuming they are always present.\n */\n\n// ============================================================================\n// TTS — POST /api/v3/tts/create\n// ============================================================================\n\n/** Output container/codec accepted by `audio_config.format`. */\nexport type BytePlusTTSAudioFormat = 'wav' | 'mp3' | 'pcm' | 'ogg_opus'\n\n/**\n * Sample rates `audio_config.sample_rate` accepts.\n *\n * The docs also state a *default* of 40000, which is not one of the valid\n * values — a documentation bug. The adapter therefore always sends an\n * explicit rate rather than relying on the server default.\n */\nexport const BYTEPLUS_TTS_SAMPLE_RATES = [\n 8000, 16000, 24000, 32000, 44100, 48000,\n] as const\n\n/** A sample rate `audio_config.sample_rate` accepts. */\nexport type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number]\n\n/**\n * One entry of the request's `references` array.\n *\n * This is where the voice lives: `speaker` names a stock voice, while\n * `audio_url` / `audio_data` supply reference clips to clone (the `@Audio1`..\n * `@Audio3` markers in `text_prompt` address them positionally). Exactly one\n * of `speaker`, `audio_url` or `audio_data` may be set per entry; at most 3\n * audio references (30 s / 10 MB each) and 1 image reference (10 MB), and\n * image references are mutually exclusive with audio ones.\n *\n * **Member object shape is unresolved — must live-probe when the Seed Speech\n * key lands.** The docs list the member fields flat (`speaker | audio_data |\n * audio_url | image_data | image_url`) without a worked example, so whether\n * the server wants a flat `{ speaker }` or a discriminated `{ type, ... }`\n * could not be settled. The adapter sends the flat reading — see\n * `buildTTSRequestBody` in `../adapters/tts`.\n */\nexport interface BytePlusTTSReference {\n /** Stock voice id, e.g. `en_female_stokie_uranus_bigtts`. */\n speaker?: string\n /** URL of a reference clip to clone (≤30 s, ≤10 MB). */\n audio_url?: string\n /** Base64 reference clip to clone (≤30 s, ≤10 MB). */\n audio_data?: string\n /** URL of a reference image (≤10 MB). Mutually exclusive with audio refs. */\n image_url?: string\n /** Base64 reference image (≤10 MB). Mutually exclusive with audio refs. */\n image_data?: string\n}\n\n/**\n * `audio_config` block of a TTS request — exactly six fields.\n *\n * The three `*_rate` fields are integer percentages relative to the voice's\n * neutral delivery, not multipliers.\n */\nexport interface BytePlusTTSAudioConfig {\n /** Output format. Defaults to `wav` server-side. */\n format?: BytePlusTTSAudioFormat\n /**\n * Output sample rate in Hz. See {@link BYTEPLUS_TTS_SAMPLE_RATES} for the\n * valid values and for why the adapter always sends one.\n */\n sample_rate?: number\n /** Speaking rate, `-50`..`100`. `-50` = 0.5×, `0` = 1×, `100` = 2×. */\n speech_rate?: number\n /** Loudness adjustment, `-50`..`100`. `0` is the voice's natural level. */\n loudness_rate?: number\n /** Pitch adjustment, `-12`..`12`. `0` is the voice's natural pitch. */\n pitch_rate?: number\n /** Emit sentence and word timings in the response. Defaults to `false`. */\n enable_subtitle?: boolean\n}\n\n/** Request body for `POST /api/v3/tts/create` — exactly five fields. */\nexport interface BytePlusTTSCreateRequest {\n /** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */\n model: string\n /**\n * The text to speak (≤3000 chars). Dual-purpose: either literal text or a\n * natural-language description of the delivery, and the place the\n * `@Audio1`..`@Audio3` reference markers go.\n *\n * **`text_prompt` is correct — do not \"fix\" this to `text`.** This endpoint\n * has no request-side `text` field. The `text` spelling belongs to the\n * *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits under\n * `req_params.text`. Confirmed against\n * docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01, which lists this\n * body as exactly `model`, `text_prompt`, `references`, `audio_config`,\n * `watermark`.\n */\n text_prompt: string\n /** Voice selection and cloning references. See {@link BytePlusTTSReference}. */\n references?: Array<BytePlusTTSReference>\n audio_config?: BytePlusTTSAudioConfig\n /**\n * Watermark the generated audio. The field name is confirmed; the boolean\n * type is assumed by analogy with Seedream's `watermark` and unprobed.\n */\n watermark?: boolean\n}\n\n/**\n * One timed entry of a TTS subtitle track.\n *\n * **Times are milliseconds** — unlike the response's `duration` fields, which\n * are seconds. The endpoint genuinely mixes units.\n */\nexport interface BytePlusTTSSubtitleEntry {\n text?: string\n start_time?: number\n end_time?: number\n}\n\n/** Sentence- and word-level timings returned when `enable_subtitle` is set. */\nexport interface BytePlusTTSSubtitle {\n sentences?: Array<BytePlusTTSSubtitleEntry>\n words?: Array<BytePlusTTSSubtitleEntry>\n}\n\n/** Response body for `POST /api/v3/tts/create`. */\nexport interface BytePlusTTSCreateResponse {\n /**\n * Status code — `0` on success, a flat error code otherwise.\n *\n * Typed as `number | string` because only the *error* envelope was verified\n * live (HTTP 401 `{\"code\":45000010,…}`); the success envelope's shape is\n * docs-derived and no voice key was available to confirm it. Treating a\n * string code as \"not a failure\" would return a failed 200 as success, so\n * the adapter coerces before comparing — see `isZeroCode` in\n * `adapters/tts.ts`.\n */\n code?: number | string\n message?: string\n /** Base64-encoded audio in the requested `audio_config.format`. */\n audio?: string\n /**\n * Length of the delivered audio in **seconds** (float), after `speech_rate`\n * is applied. This can legitimately exceed 120 when the clip is slowed\n * down — the 120 s cap applies to {@link BytePlusTTSCreateResponse.original_duration}.\n */\n duration?: number | string\n /**\n * Length in **seconds** (float) before rate adjustment. This is the billing\n * basis and is capped at 120.\n */\n original_duration?: number | string\n /** Temporary download URL for the same audio. Expires after ~2 hours. */\n url?: string\n /** Sentence and word timings, present when `enable_subtitle` was set. */\n subtitle?: BytePlusTTSSubtitle\n}\n\n// ============================================================================\n// ASR — POST /api/v3/auc/bigmodel/recognize/flash\n// ============================================================================\n\n/**\n * Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo\n * model. The flash endpoint takes no `model` field in its body — the model is\n * chosen entirely by this header.\n */\nexport const BYTEPLUS_ASR_RESOURCE_ID = 'volc.seedasr.auc_turbo'\n\n/** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */\nexport const BYTEPLUS_ASR_RESOURCE_HEADER = 'X-Api-Resource-Id'\n\n/**\n * Audio input. Exactly one of `url` or `data` is sent — the endpoint accepts\n * files up to 2 hours long / 100 MB.\n */\nexport interface BytePlusASRAudio {\n /** Publicly reachable URL of the audio file. */\n url?: string\n /** Base64-encoded audio bytes. */\n data?: string\n /** Container hint, e.g. `mp3`, `wav`, `ogg`. */\n format?: string\n}\n\n/** `request` block of a recognition call. */\nexport interface BytePlusASRRequestOptions {\n /** Recognition model family. Defaults to `bigmodel`. */\n model_name?: string\n /** Inverse text normalisation (spoken numbers → digits). */\n enable_itn?: boolean\n /** Insert punctuation. */\n enable_punc?: boolean\n /** Disfluency removal (\"um\", repeated words). */\n enable_ddc?: boolean\n /** Attach per-utterance speaker labels. */\n enable_speaker_info?: boolean\n /** Return the `utterances` breakdown as well as the flat transcript. */\n show_utterances?: boolean\n /** Spoken language hint, e.g. `en-US`. */\n language?: string\n}\n\n/** Request body for `POST /api/v3/auc/bigmodel/recognize/flash`. */\nexport interface BytePlusASRRecognizeRequest {\n user?: { uid?: string }\n audio: BytePlusASRAudio\n request?: BytePlusASRRequestOptions\n}\n\n/** One recognised word. `start_time` / `end_time` are milliseconds. */\nexport interface BytePlusASRWord {\n text?: string\n start_time?: number\n end_time?: number\n confidence?: number\n}\n\n/** One recognised utterance. `start_time` / `end_time` are milliseconds. */\nexport interface BytePlusASRUtterance {\n text?: string\n start_time?: number\n end_time?: number\n words?: Array<BytePlusASRWord>\n /**\n * Extra per-utterance annotations. Speaker labels arrive here when\n * `enable_speaker_info` is set; the exact key is read defensively because it\n * could not be confirmed against a live response.\n */\n additions?: Record<string, string>\n}\n\nexport interface BytePlusASRResult {\n text?: string\n utterances?: Array<BytePlusASRUtterance>\n}\n\n/**\n * Response body for `POST /api/v3/auc/bigmodel/recognize/flash`.\n *\n * The Volcengine-lineage wire shape nests everything under `result`; BytePlus'\n * prose docs describe the same payload as \"transcript + utterances\", so the\n * flat spelling is tolerated as a fallback.\n */\nexport interface BytePlusASRRecognizeResponse {\n /** `duration` is the audio length in **milliseconds**. */\n audio_info?: { duration?: number }\n result?: BytePlusASRResult\n /** Flat alias for `result.text`. */\n transcript?: string\n /** Flat alias for `result.utterances`. */\n utterances?: Array<BytePlusASRUtterance>\n}\n\n// ============================================================================\n// Errors\n// ============================================================================\n\n/**\n * Seed Speech error envelope: a flat numeric `code` plus a `message`, e.g.\n * `{\"code\": 45000010, \"message\": \"Invalid X-Api-Key\"}` (verified live on a\n * 401). Format it with `bytePlusVoiceError` from `../utils/client`.\n */\nexport interface BytePlusVoiceErrorBody {\n code?: number\n message?: string\n}\n"],"mappings":";;;;;;;;AAyCA,IAAa,4BAA4B;CACvC;CAAM;CAAO;CAAO;CAAO;CAAO;AACpC;;;;;;AAmJA,IAAa,2BAA2B;;AAGxC,IAAa,+BAA+B"}
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
import { BytePlusImageOutputFormat, BytePlusImageResponseFormat, BytePlusOptimizePromptOptions, BytePlusSequentialImageGeneration, BytePlusSequentialImageGenerationOptions } from './wire-types.js';
|
|
2
|
+
import { BytePlusImageModel, BytePlusImageSize } from '../model-meta.js';
|
|
3
|
+
/**
|
|
4
|
+
* BytePlus documents a 600-word ceiling on the image prompt. Word-based, so
|
|
5
|
+
* it is only meaningful for space-separated scripts — the check below never
|
|
6
|
+
* fires for Chinese or Japanese text, which is the intended behaviour.
|
|
7
|
+
*/
|
|
8
|
+
export declare const BYTEPLUS_IMAGE_MAX_PROMPT_WORDS = 600;
|
|
9
|
+
/**
|
|
10
|
+
* Upper bound of `sequential_image_generation_options.max_images`, i.e. the
|
|
11
|
+
* most images one request can return.
|
|
12
|
+
*/
|
|
13
|
+
export declare const BYTEPLUS_IMAGE_MAX_SEQUENTIAL_IMAGES = 15;
|
|
14
|
+
/**
|
|
15
|
+
* Models that accept `output_format`.
|
|
16
|
+
*
|
|
17
|
+
* The Ark OpenAPI document's field note claims 5.0-lite only, but its own
|
|
18
|
+
* request demo sends `output_format` on `seedream-5-0-260128`, so the whole
|
|
19
|
+
* 5.0 family is treated as supporting it. The Seedream 4.x snapshots are
|
|
20
|
+
* documented as not reading it, so `output_format` is omitted from their
|
|
21
|
+
* provider-options type — but, as with `sequential_image_generation`, it is
|
|
22
|
+
* not gated at runtime: a value that reaches a model which does not read it
|
|
23
|
+
* comes back as an Ark error rather than a local rejection.
|
|
24
|
+
*/
|
|
25
|
+
export declare const BYTEPLUS_OUTPUT_FORMAT_IMAGE_MODELS: ReadonlyArray<BytePlusImageModel>;
|
|
26
|
+
/**
|
|
27
|
+
* Base provider options shared by every Seedream model.
|
|
28
|
+
*/
|
|
29
|
+
export interface BytePlusImageBaseProviderOptions {
|
|
30
|
+
/**
|
|
31
|
+
* Return images as expiring links (`url`, valid 24 hours) or inline base64
|
|
32
|
+
* (`b64_json`).
|
|
33
|
+
*
|
|
34
|
+
* @default 'url'
|
|
35
|
+
*/
|
|
36
|
+
response_format?: BytePlusImageResponseFormat;
|
|
37
|
+
/**
|
|
38
|
+
* Whether to stamp an "AI generated" watermark in the bottom-right corner.
|
|
39
|
+
*
|
|
40
|
+
* **BytePlus defaults this to `true`.** Pass `false` for a clean image.
|
|
41
|
+
*/
|
|
42
|
+
watermark?: boolean;
|
|
43
|
+
/**
|
|
44
|
+
* Group-image mode. Set to `auto` to let the model return a set of related
|
|
45
|
+
* images (bounded by {@link BytePlusImageBaseProviderOptions.sequential_image_generation_options}).
|
|
46
|
+
* `generateImage()`'s `numberOfImages` sets this for you; an explicit value
|
|
47
|
+
* here wins.
|
|
48
|
+
*
|
|
49
|
+
* Documented on Seedream 5.0-lite, 4.5 and 4.0. It is sent as given on
|
|
50
|
+
* every model rather than gated locally — the shipped 5.0 ids post-date the
|
|
51
|
+
* published parameter table, and an unsupported combination comes back as a
|
|
52
|
+
* clear Ark error.
|
|
53
|
+
*
|
|
54
|
+
* @default 'disabled'
|
|
55
|
+
*/
|
|
56
|
+
sequential_image_generation?: BytePlusSequentialImageGeneration;
|
|
57
|
+
/** Bounds for group-image mode. Only read when the mode is `auto`. */
|
|
58
|
+
sequential_image_generation_options?: BytePlusSequentialImageGenerationOptions;
|
|
59
|
+
/**
|
|
60
|
+
* Prompt-rewriting configuration. Documented on Seedream 5.0-lite, 4.5 and
|
|
61
|
+
* 4.0; `mode: 'fast'` is unsupported on 5.0-lite and 4.5.
|
|
62
|
+
*/
|
|
63
|
+
optimize_prompt_options?: BytePlusOptimizePromptOptions;
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* Provider options for the Seedream 5.0 family, which additionally chooses the
|
|
67
|
+
* generated file format.
|
|
68
|
+
*/
|
|
69
|
+
export interface BytePlusSeedream5ImageProviderOptions extends BytePlusImageBaseProviderOptions {
|
|
70
|
+
/**
|
|
71
|
+
* File format of the generated image.
|
|
72
|
+
*
|
|
73
|
+
* @default 'jpeg'
|
|
74
|
+
*/
|
|
75
|
+
output_format?: BytePlusImageOutputFormat;
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Every Seedream provider option, used as the adapter's base option type.
|
|
79
|
+
* Call sites are narrowed per model by
|
|
80
|
+
* {@link BytePlusImageModelProviderOptionsByName}.
|
|
81
|
+
*/
|
|
82
|
+
export type BytePlusImageProviderOptions = BytePlusSeedream5ImageProviderOptions;
|
|
83
|
+
/**
|
|
84
|
+
* Type-only map from image model name to its provider options.
|
|
85
|
+
*/
|
|
86
|
+
export type BytePlusImageModelProviderOptionsByName = {
|
|
87
|
+
'dola-seedream-5-0-pro-260628': BytePlusSeedream5ImageProviderOptions;
|
|
88
|
+
'seedream-5-0-260128': BytePlusSeedream5ImageProviderOptions;
|
|
89
|
+
'seedream-5-0-lite-260128': BytePlusSeedream5ImageProviderOptions;
|
|
90
|
+
'seedream-4-5-251128': BytePlusImageBaseProviderOptions;
|
|
91
|
+
'seedream-4-0-250828': BytePlusImageBaseProviderOptions;
|
|
92
|
+
};
|
|
93
|
+
/**
|
|
94
|
+
* Type-only map from image model name to the non-text prompt modalities it
|
|
95
|
+
* accepts. Every shipped Seedream model takes reference images for
|
|
96
|
+
* image-conditioned generation.
|
|
97
|
+
*/
|
|
98
|
+
export type BytePlusImageModelInputModalitiesByName = {
|
|
99
|
+
[K in BytePlusImageModel]: readonly ['image'];
|
|
100
|
+
};
|
|
101
|
+
/**
|
|
102
|
+
* A parsed `size` value: either the shorthand token form or explicit pixels.
|
|
103
|
+
*/
|
|
104
|
+
export type ParsedBytePlusImageSize = {
|
|
105
|
+
kind: 'token';
|
|
106
|
+
value: '1K' | '2K' | '4K';
|
|
107
|
+
} | {
|
|
108
|
+
kind: 'pixels';
|
|
109
|
+
width: number;
|
|
110
|
+
height: number;
|
|
111
|
+
};
|
|
112
|
+
/**
|
|
113
|
+
* Parses a Seedream `size` string. Accepts a shorthand token (case-insensitive
|
|
114
|
+
* — `2k` normalizes to `2K`) or explicit `WIDTHxHEIGHT` pixels, and returns
|
|
115
|
+
* `undefined` for anything else, including mixtures such as `2K x 1024`.
|
|
116
|
+
*/
|
|
117
|
+
export declare function parseBytePlusImageSize(size: string): ParsedBytePlusImageSize | undefined;
|
|
118
|
+
/**
|
|
119
|
+
* Validates the generic `size` option and returns the string to put on the
|
|
120
|
+
* wire (`2K`, `2048x2048`), or `undefined` when no size was requested.
|
|
121
|
+
*
|
|
122
|
+
* This checks the *form* only. Which pixel dimensions a given model actually
|
|
123
|
+
* accepts is not encoded here, so the message deliberately makes no per-model
|
|
124
|
+
* claim; an out-of-range size is left to the API to reject.
|
|
125
|
+
*
|
|
126
|
+
* @throws Error when the value is neither a size token nor `WIDTHxHEIGHT`.
|
|
127
|
+
*/
|
|
128
|
+
export declare function resolveBytePlusImageSize(size: BytePlusImageSize | string | undefined): string | undefined;
|
|
129
|
+
/**
|
|
130
|
+
* Validates the prompt text against BytePlus's documented limits.
|
|
131
|
+
*
|
|
132
|
+
* @throws Error when the prompt is empty or exceeds
|
|
133
|
+
* {@link BYTEPLUS_IMAGE_MAX_PROMPT_WORDS} words.
|
|
134
|
+
*/
|
|
135
|
+
export declare function validateBytePlusImagePrompt(model: string, prompt: string): void;
|
|
136
|
+
/**
|
|
137
|
+
* Validates the reference-image count against the model's editing limit.
|
|
138
|
+
*
|
|
139
|
+
* A model this package has no limit for is left to Ark, deliberately and
|
|
140
|
+
* explicitly. `model` is typed closed, but `wire-types.ts` documents that the
|
|
141
|
+
* endpoint also accepts preconfigured endpoint ids (`ep-…`), so a JS caller
|
|
142
|
+
* can reach an id that is not in the table. Reading `undefined` out of it and
|
|
143
|
+
* comparing `count > undefined` — always false — would disable the guard by
|
|
144
|
+
* accident and look identical to passing; the explicit early return says the
|
|
145
|
+
* skip is intended.
|
|
146
|
+
*
|
|
147
|
+
* @throws Error when more references are supplied than a *known* model accepts.
|
|
148
|
+
*/
|
|
149
|
+
export declare function validateBytePlusReferenceImages(model: BytePlusImageModel, count: number): void;
|
|
150
|
+
/**
|
|
151
|
+
* Maps the generic `numberOfImages` option onto Seedream's group-image
|
|
152
|
+
* parameters.
|
|
153
|
+
*
|
|
154
|
+
* The endpoint has no `n`: more than one image per request is only reachable
|
|
155
|
+
* through `sequential_image_generation: 'auto'`, where `max_images` is an
|
|
156
|
+
* upper bound and the model decides how many images the prompt actually
|
|
157
|
+
* warrants. A request for N images can therefore come back with fewer — the
|
|
158
|
+
* one place BytePlus cannot honour `numberOfImages` exactly.
|
|
159
|
+
*
|
|
160
|
+
* @throws Error when the count is not an integer in `[1, 15]`.
|
|
161
|
+
*/
|
|
162
|
+
export declare function resolveBytePlusSequentialImages(model: string, numberOfImages: number | undefined): {
|
|
163
|
+
sequential_image_generation?: BytePlusSequentialImageGeneration;
|
|
164
|
+
sequential_image_generation_options?: BytePlusSequentialImageGenerationOptions;
|
|
165
|
+
};
|