@tanstack/ai-byteplus 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +202 -0
- package/dist/esm/adapters/image.d.ts +89 -0
- package/dist/esm/adapters/image.js +229 -0
- package/dist/esm/adapters/image.js.map +1 -0
- package/dist/esm/adapters/text.d.ts +163 -0
- package/dist/esm/adapters/text.js +347 -0
- package/dist/esm/adapters/text.js.map +1 -0
- package/dist/esm/adapters/transcription.d.ts +102 -0
- package/dist/esm/adapters/transcription.js +274 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +143 -0
- package/dist/esm/adapters/tts.js +307 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/adapters/video.d.ts +182 -0
- package/dist/esm/adapters/video.js +442 -0
- package/dist/esm/adapters/video.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +46 -0
- package/dist/esm/audio/tts-provider-options.d.ts +114 -0
- package/dist/esm/audio/wire-types.d.ts +261 -0
- package/dist/esm/audio/wire-types.js +28 -0
- package/dist/esm/audio/wire-types.js.map +1 -0
- package/dist/esm/image/image-provider-options.d.ts +165 -0
- package/dist/esm/image/image-provider-options.js +134 -0
- package/dist/esm/image/image-provider-options.js.map +1 -0
- package/dist/esm/image/wire-types.d.ts +149 -0
- package/dist/esm/index.d.ts +25 -0
- package/dist/esm/index.js +11 -0
- package/dist/esm/message-types.d.ts +154 -0
- package/dist/esm/model-meta.d.ts +594 -0
- package/dist/esm/model-meta.js +619 -0
- package/dist/esm/model-meta.js.map +1 -0
- package/dist/esm/text/text-provider-options.d.ts +109 -0
- package/dist/esm/utils/client.d.ts +183 -0
- package/dist/esm/utils/client.js +253 -0
- package/dist/esm/utils/client.js.map +1 -0
- package/dist/esm/video/video-provider-options.d.ts +197 -0
- package/dist/esm/video/video-provider-options.js +191 -0
- package/dist/esm/video/video-provider-options.js.map +1 -0
- package/dist/esm/video/wire-types.d.ts +248 -0
- package/package.json +77 -0
- package/src/adapters/image.ts +409 -0
- package/src/adapters/text.ts +539 -0
- package/src/adapters/transcription.ts +479 -0
- package/src/adapters/tts.ts +447 -0
- package/src/adapters/video.ts +732 -0
- package/src/audio/transcription-provider-options.ts +46 -0
- package/src/audio/tts-provider-options.ts +122 -0
- package/src/audio/wire-types.ts +290 -0
- package/src/image/image-provider-options.ts +288 -0
- package/src/image/wire-types.ts +169 -0
- package/src/index.ts +222 -0
- package/src/message-types.ts +169 -0
- package/src/model-meta.ts +954 -0
- package/src/text/text-provider-options.ts +151 -0
- package/src/utils/client.ts +377 -0
- package/src/video/video-provider-options.ts +361 -0
- package/src/video/wire-types.ts +293 -0
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provider-specific options for BytePlus Seed Speech ASR
|
|
3
|
+
* (`POST /api/v3/auc/bigmodel/recognize/flash`).
|
|
4
|
+
*
|
|
5
|
+
* Field names mirror the wire format: everything except `uid` and
|
|
6
|
+
* `audio_format` is forwarded inside the request's `request` block.
|
|
7
|
+
*/
|
|
8
|
+
export interface BytePlusTranscriptionProviderOptions {
|
|
9
|
+
/**
|
|
10
|
+
* Recognition model family. Defaults to `bigmodel`; the specific model is
|
|
11
|
+
* selected by the `X-Api-Resource-Id` header rather than by this value.
|
|
12
|
+
*/
|
|
13
|
+
model_name?: string
|
|
14
|
+
/**
|
|
15
|
+
* Container hint for the submitted audio (`mp3`, `wav`, `ogg`, …). Only
|
|
16
|
+
* needed when the format can't be inferred from the input — a `File`'s name
|
|
17
|
+
* or MIME type, or a data URL's MIME type, is used automatically.
|
|
18
|
+
*/
|
|
19
|
+
audio_format?: string
|
|
20
|
+
/** Inverse text normalisation: render spoken numbers, dates etc. as digits. */
|
|
21
|
+
enable_itn?: boolean
|
|
22
|
+
/** Insert punctuation into the transcript. */
|
|
23
|
+
enable_punc?: boolean
|
|
24
|
+
/** Disfluency removal — drop fillers and stutters. */
|
|
25
|
+
enable_ddc?: boolean
|
|
26
|
+
/**
|
|
27
|
+
* Attach speaker labels to each utterance. When present they are surfaced
|
|
28
|
+
* as `segment.speaker` on the result.
|
|
29
|
+
*/
|
|
30
|
+
enable_speaker_info?: boolean
|
|
31
|
+
/**
|
|
32
|
+
* Return the per-utterance breakdown as well as the flat transcript.
|
|
33
|
+
* Defaults to `true` so `segments` and `words` are populated.
|
|
34
|
+
*/
|
|
35
|
+
show_utterances?: boolean
|
|
36
|
+
/**
|
|
37
|
+
* Spoken-language hint (e.g. `en-US`). Overrides the cross-provider
|
|
38
|
+
* `TranscriptionOptions.language`.
|
|
39
|
+
*/
|
|
40
|
+
language?: string
|
|
41
|
+
/**
|
|
42
|
+
* Caller identifier echoed back in BytePlus' request logs. Useful for
|
|
43
|
+
* correlating usage; defaults to `tanstack-ai`.
|
|
44
|
+
*/
|
|
45
|
+
uid?: string
|
|
46
|
+
}
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
BytePlusTTSAudioFormat,
|
|
3
|
+
BytePlusTTSReference,
|
|
4
|
+
BytePlusTTSSampleRate,
|
|
5
|
+
BytePlusTTSSubtitle,
|
|
6
|
+
} from './wire-types'
|
|
7
|
+
import type { TTSResult } from '@tanstack/ai'
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Seed Speech voice identifier (`speaker` on the wire).
|
|
11
|
+
*
|
|
12
|
+
* Voice ids encode language, gender, character name and model generation:
|
|
13
|
+
* `en_female_stokie_uranus_bigtts` is the English female "Stokie" voice on
|
|
14
|
+
* TTS 2.0. The generation suffix matters when picking one:
|
|
15
|
+
*
|
|
16
|
+
* - `_uranus_bigtts` — TTS 2.0 voices (the current generation).
|
|
17
|
+
* - `_mars_bigtts` / `_moon_bigtts` — TTS 1.0 voices.
|
|
18
|
+
* - `*_emo_v2_*` — TTS 1.0 voices that additionally accept emotion tags.
|
|
19
|
+
*
|
|
20
|
+
* The full roster lives at
|
|
21
|
+
* https://docs.byteplus.com/en/docs/byteplusvoice/voicelist and changes far
|
|
22
|
+
* more often than this package ships, so the union stays open: any string is
|
|
23
|
+
* accepted, and the one listed id is the adapter's default.
|
|
24
|
+
*/
|
|
25
|
+
export type BytePlusTTSVoice = 'en_female_stokie_uranus_bigtts' | (string & {})
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Provider-specific options for BytePlus Seed Speech TTS
|
|
29
|
+
* (`POST /api/v3/tts/create`).
|
|
30
|
+
*
|
|
31
|
+
* These map 1:1 onto the wire fields so the BytePlus documentation stays
|
|
32
|
+
* useful; where a cross-provider `TTSOptions` field covers the same ground
|
|
33
|
+
* (`voice`, `format`, `speed`), the option here wins.
|
|
34
|
+
*/
|
|
35
|
+
export interface BytePlusTTSProviderOptions {
|
|
36
|
+
/**
|
|
37
|
+
* Voice id. Overrides `TTSOptions.voice` when both are set. It is sent as
|
|
38
|
+
* the `speaker` of a single `references` entry.
|
|
39
|
+
*/
|
|
40
|
+
speaker?: BytePlusTTSVoice
|
|
41
|
+
/**
|
|
42
|
+
* Full `references` array, for voice cloning or image-referenced delivery.
|
|
43
|
+
* When set this replaces the entry the adapter would otherwise build from
|
|
44
|
+
* `speaker` / `TTSOptions.voice`, so include a `speaker` entry yourself if
|
|
45
|
+
* you still want a stock voice. Address audio references from the text with
|
|
46
|
+
* `@Audio1`..`@Audio3`.
|
|
47
|
+
*/
|
|
48
|
+
references?: Array<BytePlusTTSReference>
|
|
49
|
+
/**
|
|
50
|
+
* Output format. Overrides the mapping applied to `TTSOptions.format`,
|
|
51
|
+
* which is useful for `ogg_opus` and for pinning `pcm` explicitly.
|
|
52
|
+
*/
|
|
53
|
+
format?: BytePlusTTSAudioFormat
|
|
54
|
+
/**
|
|
55
|
+
* Output sample rate in Hz. Defaults to 24000 — the adapter always sends an
|
|
56
|
+
* explicit rate because the documented server default (40000) is not one of
|
|
57
|
+
* the values the endpoint accepts. For `pcm` output this is also what the
|
|
58
|
+
* returned `contentType` (`audio/L16;rate=…`) reports.
|
|
59
|
+
*/
|
|
60
|
+
sample_rate?: BytePlusTTSSampleRate
|
|
61
|
+
/**
|
|
62
|
+
* Pitch adjustment in the range `-12`..`12`, where `0` is the voice's
|
|
63
|
+
* natural pitch.
|
|
64
|
+
*/
|
|
65
|
+
pitch_rate?: number
|
|
66
|
+
/**
|
|
67
|
+
* Speaking rate in the range `-50`..`100` (`-50` = 0.5×, `0` = 1×,
|
|
68
|
+
* `100` = 2×). Overrides the value derived from `TTSOptions.speed`.
|
|
69
|
+
*/
|
|
70
|
+
speech_rate?: number
|
|
71
|
+
/**
|
|
72
|
+
* Loudness adjustment in the range `-50`..`100`, where `0` is the voice's
|
|
73
|
+
* natural level.
|
|
74
|
+
*/
|
|
75
|
+
loudness_rate?: number
|
|
76
|
+
/**
|
|
77
|
+
* Ask for sentence- and word-level timings alongside the audio. They are
|
|
78
|
+
* surfaced on {@link BytePlusTTSResult.subtitle}.
|
|
79
|
+
*/
|
|
80
|
+
enable_subtitle?: boolean
|
|
81
|
+
/**
|
|
82
|
+
* Watermark the generated audio. The field name is confirmed against the
|
|
83
|
+
* endpoint schema; the boolean type is assumed and unprobed.
|
|
84
|
+
*/
|
|
85
|
+
watermark?: boolean
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* BytePlus-specific extension of `TTSResult`.
|
|
90
|
+
*
|
|
91
|
+
* The cross-provider `TTSResult` has nowhere to put the subtitle timings or
|
|
92
|
+
* the temporary download URL, so callers who want them narrow the result:
|
|
93
|
+
*
|
|
94
|
+
* ```ts
|
|
95
|
+
* const result: BytePlusTTSResult = await generateSpeech({ adapter, text })
|
|
96
|
+
* for (const sentence of result.subtitle?.sentences ?? []) {
|
|
97
|
+
* console.log(sentence.text, sentence.start_time)
|
|
98
|
+
* }
|
|
99
|
+
* ```
|
|
100
|
+
*/
|
|
101
|
+
export interface BytePlusTTSResult extends TTSResult {
|
|
102
|
+
/**
|
|
103
|
+
* Sentence and word timings, present only when
|
|
104
|
+
* `modelOptions.enable_subtitle` was set. Their `start_time` / `end_time`
|
|
105
|
+
* are **milliseconds**, even though `duration` and
|
|
106
|
+
* {@link BytePlusTTSResult.originalDuration} are seconds.
|
|
107
|
+
*/
|
|
108
|
+
subtitle?: BytePlusTTSSubtitle
|
|
109
|
+
/**
|
|
110
|
+
* Length of the audio in seconds *before* `speech_rate` was applied. This
|
|
111
|
+
* is what BytePlus bills on and what the 120 s cap applies to, so it is the
|
|
112
|
+
* number to meter against — `duration` reflects the delivered clip and can
|
|
113
|
+
* exceed 120 s when the speech is slowed down.
|
|
114
|
+
*/
|
|
115
|
+
originalDuration?: number
|
|
116
|
+
/**
|
|
117
|
+
* Temporary download URL for the same audio that `audio` carries as base64.
|
|
118
|
+
* **Expires roughly 2 hours after generation** — persist the bytes, not the
|
|
119
|
+
* link.
|
|
120
|
+
*/
|
|
121
|
+
url?: string
|
|
122
|
+
}
|
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Minimal wire types for the BytePlus **Seed Speech** HTTP API (TTS + ASR).
|
|
3
|
+
*
|
|
4
|
+
* Seed Speech is a separate product from Ark: it lives on
|
|
5
|
+
* `voice.ap-southeast-1.bytepluses.com`, authenticates with `X-Api-Key`
|
|
6
|
+
* (a different key from `ARK_API_KEY`), and returns a flat numeric error
|
|
7
|
+
* envelope instead of Ark's OpenAI-shaped one.
|
|
8
|
+
*
|
|
9
|
+
* Only the fields the adapters read or write are modelled here — this is a
|
|
10
|
+
* hand-written subset, not a generated schema.
|
|
11
|
+
*
|
|
12
|
+
* Provenance:
|
|
13
|
+
* - Endpoints, auth header, format/rate ranges and the 120 s TTS output cap:
|
|
14
|
+
* BytePlus Seed Speech docs (`docs.byteplus.com/en/docs/byteplusvoice`),
|
|
15
|
+
* captured in the Phase 0 research notes.
|
|
16
|
+
* - Error envelope `{code, message}`: verified live — an Ark key sent as
|
|
17
|
+
* `X-Api-Key` returns HTTP 401 `{"code":45000010,"message":"Invalid X-Api-Key"}`.
|
|
18
|
+
* - ASR request/response shape (`user`/`audio`/`request` in, `audio_info` +
|
|
19
|
+
* `result.utterances` out, all timings in **milliseconds**): the Volcengine
|
|
20
|
+
* flash-recognition reference the BytePlus endpoint is derived from
|
|
21
|
+
* (`docs.volcengine.com/docs/6561/1631584`).
|
|
22
|
+
*
|
|
23
|
+
* No Seed Speech API key was available when these were written, so the TTS
|
|
24
|
+
* response fields are documented-but-unverified; the adapters parse them
|
|
25
|
+
* defensively rather than assuming they are always present.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
// ============================================================================
|
|
29
|
+
// TTS — POST /api/v3/tts/create
|
|
30
|
+
// ============================================================================
|
|
31
|
+
|
|
32
|
+
/** Output container/codec accepted by `audio_config.format`. */
|
|
33
|
+
export type BytePlusTTSAudioFormat = 'wav' | 'mp3' | 'pcm' | 'ogg_opus'
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Sample rates `audio_config.sample_rate` accepts.
|
|
37
|
+
*
|
|
38
|
+
* The docs also state a *default* of 40000, which is not one of the valid
|
|
39
|
+
* values — a documentation bug. The adapter therefore always sends an
|
|
40
|
+
* explicit rate rather than relying on the server default.
|
|
41
|
+
*/
|
|
42
|
+
export const BYTEPLUS_TTS_SAMPLE_RATES = [
|
|
43
|
+
8000, 16000, 24000, 32000, 44100, 48000,
|
|
44
|
+
] as const
|
|
45
|
+
|
|
46
|
+
/** A sample rate `audio_config.sample_rate` accepts. */
|
|
47
|
+
export type BytePlusTTSSampleRate = (typeof BYTEPLUS_TTS_SAMPLE_RATES)[number]
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* One entry of the request's `references` array.
|
|
51
|
+
*
|
|
52
|
+
* This is where the voice lives: `speaker` names a stock voice, while
|
|
53
|
+
* `audio_url` / `audio_data` supply reference clips to clone (the `@Audio1`..
|
|
54
|
+
* `@Audio3` markers in `text_prompt` address them positionally). Exactly one
|
|
55
|
+
* of `speaker`, `audio_url` or `audio_data` may be set per entry; at most 3
|
|
56
|
+
* audio references (30 s / 10 MB each) and 1 image reference (10 MB), and
|
|
57
|
+
* image references are mutually exclusive with audio ones.
|
|
58
|
+
*
|
|
59
|
+
* **Member object shape is unresolved — must live-probe when the Seed Speech
|
|
60
|
+
* key lands.** The docs list the member fields flat (`speaker | audio_data |
|
|
61
|
+
* audio_url | image_data | image_url`) without a worked example, so whether
|
|
62
|
+
* the server wants a flat `{ speaker }` or a discriminated `{ type, ... }`
|
|
63
|
+
* could not be settled. The adapter sends the flat reading — see
|
|
64
|
+
* `buildTTSRequestBody` in `../adapters/tts`.
|
|
65
|
+
*/
|
|
66
|
+
export interface BytePlusTTSReference {
|
|
67
|
+
/** Stock voice id, e.g. `en_female_stokie_uranus_bigtts`. */
|
|
68
|
+
speaker?: string
|
|
69
|
+
/** URL of a reference clip to clone (≤30 s, ≤10 MB). */
|
|
70
|
+
audio_url?: string
|
|
71
|
+
/** Base64 reference clip to clone (≤30 s, ≤10 MB). */
|
|
72
|
+
audio_data?: string
|
|
73
|
+
/** URL of a reference image (≤10 MB). Mutually exclusive with audio refs. */
|
|
74
|
+
image_url?: string
|
|
75
|
+
/** Base64 reference image (≤10 MB). Mutually exclusive with audio refs. */
|
|
76
|
+
image_data?: string
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* `audio_config` block of a TTS request — exactly six fields.
|
|
81
|
+
*
|
|
82
|
+
* The three `*_rate` fields are integer percentages relative to the voice's
|
|
83
|
+
* neutral delivery, not multipliers.
|
|
84
|
+
*/
|
|
85
|
+
export interface BytePlusTTSAudioConfig {
|
|
86
|
+
/** Output format. Defaults to `wav` server-side. */
|
|
87
|
+
format?: BytePlusTTSAudioFormat
|
|
88
|
+
/**
|
|
89
|
+
* Output sample rate in Hz. See {@link BYTEPLUS_TTS_SAMPLE_RATES} for the
|
|
90
|
+
* valid values and for why the adapter always sends one.
|
|
91
|
+
*/
|
|
92
|
+
sample_rate?: number
|
|
93
|
+
/** Speaking rate, `-50`..`100`. `-50` = 0.5×, `0` = 1×, `100` = 2×. */
|
|
94
|
+
speech_rate?: number
|
|
95
|
+
/** Loudness adjustment, `-50`..`100`. `0` is the voice's natural level. */
|
|
96
|
+
loudness_rate?: number
|
|
97
|
+
/** Pitch adjustment, `-12`..`12`. `0` is the voice's natural pitch. */
|
|
98
|
+
pitch_rate?: number
|
|
99
|
+
/** Emit sentence and word timings in the response. Defaults to `false`. */
|
|
100
|
+
enable_subtitle?: boolean
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** Request body for `POST /api/v3/tts/create` — exactly five fields. */
|
|
104
|
+
export interface BytePlusTTSCreateRequest {
|
|
105
|
+
/** Seed Speech synthesis model, e.g. `seed-audio-1.0`. */
|
|
106
|
+
model: string
|
|
107
|
+
/**
|
|
108
|
+
* The text to speak (≤3000 chars). Dual-purpose: either literal text or a
|
|
109
|
+
* natural-language description of the delivery, and the place the
|
|
110
|
+
* `@Audio1`..`@Audio3` reference markers go.
|
|
111
|
+
*
|
|
112
|
+
* **`text_prompt` is correct — do not "fix" this to `text`.** This endpoint
|
|
113
|
+
* has no request-side `text` field. The `text` spelling belongs to the
|
|
114
|
+
* *other* endpoint, `/tts/unidirectional` (TTS 2.0), where it sits under
|
|
115
|
+
* `req_params.text`. Confirmed against
|
|
116
|
+
* docs.byteplus.com/en/docs/byteplusvoice/seedaudio-01, which lists this
|
|
117
|
+
* body as exactly `model`, `text_prompt`, `references`, `audio_config`,
|
|
118
|
+
* `watermark`.
|
|
119
|
+
*/
|
|
120
|
+
text_prompt: string
|
|
121
|
+
/** Voice selection and cloning references. See {@link BytePlusTTSReference}. */
|
|
122
|
+
references?: Array<BytePlusTTSReference>
|
|
123
|
+
audio_config?: BytePlusTTSAudioConfig
|
|
124
|
+
/**
|
|
125
|
+
* Watermark the generated audio. The field name is confirmed; the boolean
|
|
126
|
+
* type is assumed by analogy with Seedream's `watermark` and unprobed.
|
|
127
|
+
*/
|
|
128
|
+
watermark?: boolean
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* One timed entry of a TTS subtitle track.
|
|
133
|
+
*
|
|
134
|
+
* **Times are milliseconds** — unlike the response's `duration` fields, which
|
|
135
|
+
* are seconds. The endpoint genuinely mixes units.
|
|
136
|
+
*/
|
|
137
|
+
export interface BytePlusTTSSubtitleEntry {
|
|
138
|
+
text?: string
|
|
139
|
+
start_time?: number
|
|
140
|
+
end_time?: number
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** Sentence- and word-level timings returned when `enable_subtitle` is set. */
|
|
144
|
+
export interface BytePlusTTSSubtitle {
|
|
145
|
+
sentences?: Array<BytePlusTTSSubtitleEntry>
|
|
146
|
+
words?: Array<BytePlusTTSSubtitleEntry>
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/** Response body for `POST /api/v3/tts/create`. */
|
|
150
|
+
export interface BytePlusTTSCreateResponse {
|
|
151
|
+
/**
|
|
152
|
+
* Status code — `0` on success, a flat error code otherwise.
|
|
153
|
+
*
|
|
154
|
+
* Typed as `number | string` because only the *error* envelope was verified
|
|
155
|
+
* live (HTTP 401 `{"code":45000010,…}`); the success envelope's shape is
|
|
156
|
+
* docs-derived and no voice key was available to confirm it. Treating a
|
|
157
|
+
* string code as "not a failure" would return a failed 200 as success, so
|
|
158
|
+
* the adapter coerces before comparing — see `isZeroCode` in
|
|
159
|
+
* `adapters/tts.ts`.
|
|
160
|
+
*/
|
|
161
|
+
code?: number | string
|
|
162
|
+
message?: string
|
|
163
|
+
/** Base64-encoded audio in the requested `audio_config.format`. */
|
|
164
|
+
audio?: string
|
|
165
|
+
/**
|
|
166
|
+
* Length of the delivered audio in **seconds** (float), after `speech_rate`
|
|
167
|
+
* is applied. This can legitimately exceed 120 when the clip is slowed
|
|
168
|
+
* down — the 120 s cap applies to {@link BytePlusTTSCreateResponse.original_duration}.
|
|
169
|
+
*/
|
|
170
|
+
duration?: number | string
|
|
171
|
+
/**
|
|
172
|
+
* Length in **seconds** (float) before rate adjustment. This is the billing
|
|
173
|
+
* basis and is capped at 120.
|
|
174
|
+
*/
|
|
175
|
+
original_duration?: number | string
|
|
176
|
+
/** Temporary download URL for the same audio. Expires after ~2 hours. */
|
|
177
|
+
url?: string
|
|
178
|
+
/** Sentence and word timings, present when `enable_subtitle` was set. */
|
|
179
|
+
subtitle?: BytePlusTTSSubtitle
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
// ============================================================================
|
|
183
|
+
// ASR — POST /api/v3/auc/bigmodel/recognize/flash
|
|
184
|
+
// ============================================================================
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Value of the `X-Api-Resource-Id` header that selects the Seed ASR turbo
|
|
188
|
+
* model. The flash endpoint takes no `model` field in its body — the model is
|
|
189
|
+
* chosen entirely by this header.
|
|
190
|
+
*/
|
|
191
|
+
export const BYTEPLUS_ASR_RESOURCE_ID = 'volc.seedasr.auc_turbo'
|
|
192
|
+
|
|
193
|
+
/** Header name carrying {@link BYTEPLUS_ASR_RESOURCE_ID}. */
|
|
194
|
+
export const BYTEPLUS_ASR_RESOURCE_HEADER = 'X-Api-Resource-Id'
|
|
195
|
+
|
|
196
|
+
/**
|
|
197
|
+
* Audio input. Exactly one of `url` or `data` is sent — the endpoint accepts
|
|
198
|
+
* files up to 2 hours long / 100 MB.
|
|
199
|
+
*/
|
|
200
|
+
export interface BytePlusASRAudio {
|
|
201
|
+
/** Publicly reachable URL of the audio file. */
|
|
202
|
+
url?: string
|
|
203
|
+
/** Base64-encoded audio bytes. */
|
|
204
|
+
data?: string
|
|
205
|
+
/** Container hint, e.g. `mp3`, `wav`, `ogg`. */
|
|
206
|
+
format?: string
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/** `request` block of a recognition call. */
|
|
210
|
+
export interface BytePlusASRRequestOptions {
|
|
211
|
+
/** Recognition model family. Defaults to `bigmodel`. */
|
|
212
|
+
model_name?: string
|
|
213
|
+
/** Inverse text normalisation (spoken numbers → digits). */
|
|
214
|
+
enable_itn?: boolean
|
|
215
|
+
/** Insert punctuation. */
|
|
216
|
+
enable_punc?: boolean
|
|
217
|
+
/** Disfluency removal ("um", repeated words). */
|
|
218
|
+
enable_ddc?: boolean
|
|
219
|
+
/** Attach per-utterance speaker labels. */
|
|
220
|
+
enable_speaker_info?: boolean
|
|
221
|
+
/** Return the `utterances` breakdown as well as the flat transcript. */
|
|
222
|
+
show_utterances?: boolean
|
|
223
|
+
/** Spoken language hint, e.g. `en-US`. */
|
|
224
|
+
language?: string
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/** Request body for `POST /api/v3/auc/bigmodel/recognize/flash`. */
|
|
228
|
+
export interface BytePlusASRRecognizeRequest {
|
|
229
|
+
user?: { uid?: string }
|
|
230
|
+
audio: BytePlusASRAudio
|
|
231
|
+
request?: BytePlusASRRequestOptions
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
/** One recognised word. `start_time` / `end_time` are milliseconds. */
|
|
235
|
+
export interface BytePlusASRWord {
|
|
236
|
+
text?: string
|
|
237
|
+
start_time?: number
|
|
238
|
+
end_time?: number
|
|
239
|
+
confidence?: number
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/** One recognised utterance. `start_time` / `end_time` are milliseconds. */
|
|
243
|
+
export interface BytePlusASRUtterance {
|
|
244
|
+
text?: string
|
|
245
|
+
start_time?: number
|
|
246
|
+
end_time?: number
|
|
247
|
+
words?: Array<BytePlusASRWord>
|
|
248
|
+
/**
|
|
249
|
+
* Extra per-utterance annotations. Speaker labels arrive here when
|
|
250
|
+
* `enable_speaker_info` is set; the exact key is read defensively because it
|
|
251
|
+
* could not be confirmed against a live response.
|
|
252
|
+
*/
|
|
253
|
+
additions?: Record<string, string>
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
export interface BytePlusASRResult {
|
|
257
|
+
text?: string
|
|
258
|
+
utterances?: Array<BytePlusASRUtterance>
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/**
|
|
262
|
+
* Response body for `POST /api/v3/auc/bigmodel/recognize/flash`.
|
|
263
|
+
*
|
|
264
|
+
* The Volcengine-lineage wire shape nests everything under `result`; BytePlus'
|
|
265
|
+
* prose docs describe the same payload as "transcript + utterances", so the
|
|
266
|
+
* flat spelling is tolerated as a fallback.
|
|
267
|
+
*/
|
|
268
|
+
export interface BytePlusASRRecognizeResponse {
|
|
269
|
+
/** `duration` is the audio length in **milliseconds**. */
|
|
270
|
+
audio_info?: { duration?: number }
|
|
271
|
+
result?: BytePlusASRResult
|
|
272
|
+
/** Flat alias for `result.text`. */
|
|
273
|
+
transcript?: string
|
|
274
|
+
/** Flat alias for `result.utterances`. */
|
|
275
|
+
utterances?: Array<BytePlusASRUtterance>
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// ============================================================================
|
|
279
|
+
// Errors
|
|
280
|
+
// ============================================================================
|
|
281
|
+
|
|
282
|
+
/**
|
|
283
|
+
* Seed Speech error envelope: a flat numeric `code` plus a `message`, e.g.
|
|
284
|
+
* `{"code": 45000010, "message": "Invalid X-Api-Key"}` (verified live on a
|
|
285
|
+
* 401). Format it with `bytePlusVoiceError` from `../utils/client`.
|
|
286
|
+
*/
|
|
287
|
+
export interface BytePlusVoiceErrorBody {
|
|
288
|
+
code?: number
|
|
289
|
+
message?: string
|
|
290
|
+
}
|