@tanstack/ai 0.55.0 → 0.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +42 -16
- package/dist/esm/activities/chat/tools/tool-calls.js +1 -0
- package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
- package/dist/esm/activities/evaluate/adapter.d.ts +160 -0
- package/dist/esm/activities/evaluate/adapter.js +23 -0
- package/dist/esm/activities/evaluate/adapter.js.map +1 -0
- package/dist/esm/activities/evaluate/index.d.ts +255 -0
- package/dist/esm/activities/evaluate/index.js +317 -0
- package/dist/esm/activities/evaluate/index.js.map +1 -0
- package/dist/esm/activities/generateSpeech/adapter.d.ts +39 -1
- package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +55 -5
- package/dist/esm/activities/generateSpeech/index.js +53 -3
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/activities/generateVoice/adapter.d.ts +62 -0
- package/dist/esm/activities/generateVoice/adapter.js +23 -0
- package/dist/esm/activities/generateVoice/adapter.js.map +1 -0
- package/dist/esm/activities/generateVoice/index.d.ts +133 -0
- package/dist/esm/activities/generateVoice/index.js +184 -0
- package/dist/esm/activities/generateVoice/index.js.map +1 -0
- package/dist/esm/activities/index.d.ts +10 -4
- package/dist/esm/activities/index.js +14 -10
- package/dist/esm/activities/middleware/types.d.ts +1 -1
- package/dist/esm/client.d.ts +3 -2
- package/dist/esm/client.js +21 -3
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +4 -2
- package/dist/esm/index.js +5 -2
- package/dist/esm/middlewares/otel.js +2 -0
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/realtime/index.d.ts +1 -1
- package/dist/esm/realtime/index.js +1 -1
- package/dist/esm/realtime/index.js.map +1 -1
- package/dist/esm/types.d.ts +225 -2
- package/package.json +3 -3
- package/skills/ai-core/media-generation/SKILL.md +132 -6
- package/src/activities/chat/tools/tool-calls.ts +9 -0
- package/src/activities/evaluate/adapter.ts +212 -0
- package/src/activities/evaluate/index.ts +614 -0
- package/src/activities/generateSpeech/adapter.ts +47 -1
- package/src/activities/generateSpeech/index.ts +149 -8
- package/src/activities/generateVoice/adapter.ts +89 -0
- package/src/activities/generateVoice/index.ts +371 -0
- package/src/activities/index.ts +69 -0
- package/src/activities/middleware/types.ts +2 -0
- package/src/client.ts +35 -8
- package/src/index.ts +21 -0
- package/src/middlewares/otel.ts +2 -0
- package/src/realtime/index.ts +1 -1
- package/src/types.ts +246 -2
package/src/activities/index.ts
CHANGED
|
@@ -20,9 +20,11 @@ import type { AnyImageAdapter } from './generateImage/adapter'
|
|
|
20
20
|
import type { AnyAudioAdapter } from './generateAudio/adapter'
|
|
21
21
|
import type { AnyVideoAdapter } from './generateVideo/adapter'
|
|
22
22
|
import type { AnyTTSAdapter } from './generateSpeech/adapter'
|
|
23
|
+
import type { AnyVoiceAdapter } from './generateVoice/adapter'
|
|
23
24
|
import type { AnyTranscriptionAdapter } from './generateTranscription/adapter'
|
|
24
25
|
import type { AnyEmbeddingAdapter } from './embed/adapter'
|
|
25
26
|
import type { AnyRerankAdapter } from './rerank/adapter'
|
|
27
|
+
import type { AnyEvaluateAdapter } from './evaluate/adapter'
|
|
26
28
|
import type { AnyWorldAdapter } from './generateWorld/adapter'
|
|
27
29
|
import type { AnyLiveVideoAdapter } from './generateLiveVideo/adapter'
|
|
28
30
|
|
|
@@ -89,6 +91,46 @@ export {
|
|
|
89
91
|
type AnyRerankAdapter,
|
|
90
92
|
} from './rerank/adapter'
|
|
91
93
|
|
|
94
|
+
// ===========================
|
|
95
|
+
// Evaluate Activity
|
|
96
|
+
// ===========================
|
|
97
|
+
|
|
98
|
+
export {
|
|
99
|
+
kind as evaluateKind,
|
|
100
|
+
decide,
|
|
101
|
+
choice,
|
|
102
|
+
score,
|
|
103
|
+
boolean,
|
|
104
|
+
type EvaluateActivityOptions,
|
|
105
|
+
type EvaluateResult,
|
|
106
|
+
type EvaluateResultMeta,
|
|
107
|
+
type EvaluateProviderOptions,
|
|
108
|
+
type ChoiceAnswer,
|
|
109
|
+
type ScoreAnswer,
|
|
110
|
+
type BooleanAnswer,
|
|
111
|
+
type InferEvaluateAnswer,
|
|
112
|
+
} from './evaluate/index'
|
|
113
|
+
|
|
114
|
+
export {
|
|
115
|
+
BaseEvaluateAdapter,
|
|
116
|
+
type EvaluateAdapter,
|
|
117
|
+
type EvaluateAdapterConfig,
|
|
118
|
+
type AnyEvaluateAdapter,
|
|
119
|
+
type EvaluateOptions,
|
|
120
|
+
type EvaluateAdapterResult,
|
|
121
|
+
type EvaluateState,
|
|
122
|
+
type EvaluateInstructions,
|
|
123
|
+
type EvaluateJsonValue,
|
|
124
|
+
type WireQuestion,
|
|
125
|
+
type WireAnswer,
|
|
126
|
+
type WireChoiceQuestion,
|
|
127
|
+
type WireScoreQuestion,
|
|
128
|
+
type WireNoulQuestion,
|
|
129
|
+
type WireChoiceAnswer,
|
|
130
|
+
type WireScoreAnswer,
|
|
131
|
+
type WireNoulAnswer,
|
|
132
|
+
} from './evaluate/adapter'
|
|
133
|
+
|
|
92
134
|
// ===========================
|
|
93
135
|
// Image Activity
|
|
94
136
|
// ===========================
|
|
@@ -162,6 +204,8 @@ export { snapToDurationOption } from './generateVideo/snap'
|
|
|
162
204
|
export {
|
|
163
205
|
kind as ttsKind,
|
|
164
206
|
generateSpeech,
|
|
207
|
+
listVoices,
|
|
208
|
+
type ListVoicesActivityOptions,
|
|
165
209
|
type TTSActivityOptions,
|
|
166
210
|
type TTSActivityResult,
|
|
167
211
|
type TTSProviderOptions,
|
|
@@ -171,9 +215,30 @@ export {
|
|
|
171
215
|
BaseTTSAdapter,
|
|
172
216
|
type TTSAdapter,
|
|
173
217
|
type TTSAdapterConfig,
|
|
218
|
+
type TTSCapabilities,
|
|
174
219
|
type AnyTTSAdapter,
|
|
175
220
|
} from './generateSpeech/adapter'
|
|
176
221
|
|
|
222
|
+
// ===========================
|
|
223
|
+
// Voice Activity
|
|
224
|
+
// ===========================
|
|
225
|
+
|
|
226
|
+
export {
|
|
227
|
+
kind as voiceKind,
|
|
228
|
+
generateVoice,
|
|
229
|
+
createVoiceOptions,
|
|
230
|
+
type VoiceActivityOptions,
|
|
231
|
+
type VoiceActivityResult,
|
|
232
|
+
type VoiceProviderOptions,
|
|
233
|
+
} from './generateVoice/index'
|
|
234
|
+
|
|
235
|
+
export {
|
|
236
|
+
BaseVoiceAdapter,
|
|
237
|
+
type VoiceAdapter,
|
|
238
|
+
type VoiceAdapterConfig,
|
|
239
|
+
type AnyVoiceAdapter,
|
|
240
|
+
} from './generateVoice/adapter'
|
|
241
|
+
|
|
177
242
|
// ===========================
|
|
178
243
|
// Transcription Activity
|
|
179
244
|
// ===========================
|
|
@@ -262,9 +327,11 @@ export type AIAdapter =
|
|
|
262
327
|
| AnyAudioAdapter
|
|
263
328
|
| AnyVideoAdapter
|
|
264
329
|
| AnyTTSAdapter
|
|
330
|
+
| AnyVoiceAdapter
|
|
265
331
|
| AnyTranscriptionAdapter
|
|
266
332
|
| AnyEmbeddingAdapter
|
|
267
333
|
| AnyRerankAdapter
|
|
334
|
+
| AnyEvaluateAdapter
|
|
268
335
|
| AnyWorldAdapter
|
|
269
336
|
| AnyLiveVideoAdapter
|
|
270
337
|
|
|
@@ -276,8 +343,10 @@ export type AdapterKind =
|
|
|
276
343
|
| 'audio'
|
|
277
344
|
| 'video'
|
|
278
345
|
| 'tts'
|
|
346
|
+
| 'voice'
|
|
279
347
|
| 'transcription'
|
|
280
348
|
| 'embedding'
|
|
281
349
|
| 'rerank'
|
|
350
|
+
| 'evaluate'
|
|
282
351
|
| 'world'
|
|
283
352
|
| 'liveVideo'
|
package/src/client.ts
CHANGED
|
@@ -3,6 +3,7 @@ import type {
|
|
|
3
3
|
ImageGenerationOptions,
|
|
4
4
|
TTSOptions,
|
|
5
5
|
TranscriptionOptions,
|
|
6
|
+
VoiceGenerationOptions,
|
|
6
7
|
VideoGenerationOptions,
|
|
7
8
|
WorldGenerationOptions,
|
|
8
9
|
LiveVideoGenerationOptions,
|
|
@@ -12,6 +13,7 @@ export type GenerationKind =
|
|
|
12
13
|
| 'image'
|
|
13
14
|
| 'audio'
|
|
14
15
|
| 'tts'
|
|
16
|
+
| 'voice'
|
|
15
17
|
| 'video'
|
|
16
18
|
| 'transcription'
|
|
17
19
|
| 'world'
|
|
@@ -21,6 +23,7 @@ type GenerationInputByKind = {
|
|
|
21
23
|
image: Omit<ImageGenerationOptions, 'logger' | 'model'>
|
|
22
24
|
audio: Omit<AudioGenerationOptions, 'logger' | 'model'>
|
|
23
25
|
tts: Omit<TTSOptions, 'logger' | 'model'>
|
|
26
|
+
voice: Omit<VoiceGenerationOptions, 'logger' | 'model'>
|
|
24
27
|
video: Omit<VideoGenerationOptions, 'logger' | 'model'>
|
|
25
28
|
transcription: Omit<TranscriptionOptions, 'logger' | 'model'>
|
|
26
29
|
world: Omit<WorldGenerationOptions, 'logger' | 'model'>
|
|
@@ -38,6 +41,7 @@ const generationKinds = [
|
|
|
38
41
|
'image',
|
|
39
42
|
'audio',
|
|
40
43
|
'tts',
|
|
44
|
+
'voice',
|
|
41
45
|
'video',
|
|
42
46
|
'transcription',
|
|
43
47
|
'world',
|
|
@@ -69,6 +73,31 @@ function assertGenerationKind(kind: unknown): asserts kind is GenerationKind {
|
|
|
69
73
|
}
|
|
70
74
|
}
|
|
71
75
|
|
|
76
|
+
/**
|
|
77
|
+
* The input field(s) that identify a generation body for a kind. Most kinds
|
|
78
|
+
* have exactly one; `voice` accepts either of its two creation modes, so any
|
|
79
|
+
* one of its keys is enough.
|
|
80
|
+
*/
|
|
81
|
+
function requiredKeysForKind(kind: GenerationKind): Array<string> {
|
|
82
|
+
// Enumerated rather than defaulted so a new generation kind has to declare
|
|
83
|
+
// the field that identifies its body instead of silently inheriting
|
|
84
|
+
// `prompt`.
|
|
85
|
+
switch (kind) {
|
|
86
|
+
case 'tts':
|
|
87
|
+
return ['text']
|
|
88
|
+
case 'transcription':
|
|
89
|
+
return ['audio']
|
|
90
|
+
case 'voice':
|
|
91
|
+
return ['prompt', 'referenceAudio']
|
|
92
|
+
case 'image':
|
|
93
|
+
case 'audio':
|
|
94
|
+
case 'video':
|
|
95
|
+
case 'world':
|
|
96
|
+
case 'liveVideo':
|
|
97
|
+
return ['prompt']
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
|
|
72
101
|
function assertInputForKind(
|
|
73
102
|
kind: GenerationKind,
|
|
74
103
|
input: unknown,
|
|
@@ -77,21 +106,19 @@ function assertInputForKind(
|
|
|
77
106
|
throw new Error(`Generation ${kind} input must be an object.`)
|
|
78
107
|
}
|
|
79
108
|
|
|
80
|
-
const
|
|
81
|
-
kind === 'tts' ? 'text' : kind === 'transcription' ? 'audio' : 'prompt'
|
|
109
|
+
const requiredKeys = requiredKeysForKind(kind)
|
|
82
110
|
|
|
83
|
-
if (!hasOwnKey(input,
|
|
84
|
-
throw new Error(
|
|
111
|
+
if (!requiredKeys.some((key) => hasOwnKey(input, key))) {
|
|
112
|
+
throw new Error(
|
|
113
|
+
`Generation ${kind} input must include ${requiredKeys.join(' or ')}.`,
|
|
114
|
+
)
|
|
85
115
|
}
|
|
86
116
|
}
|
|
87
117
|
|
|
88
118
|
function isInputForKind(kind: GenerationKind, input: unknown): boolean {
|
|
89
119
|
if (!isRecord(input)) return false
|
|
90
120
|
|
|
91
|
-
|
|
92
|
-
kind === 'tts' ? 'text' : kind === 'transcription' ? 'audio' : 'prompt'
|
|
93
|
-
|
|
94
|
-
return hasOwnKey(input, requiredKey)
|
|
121
|
+
return requiredKeysForKind(kind).some((key) => hasOwnKey(input, key))
|
|
95
122
|
}
|
|
96
123
|
|
|
97
124
|
function forwardedPropsFromEnvelope(
|
package/src/index.ts
CHANGED
|
@@ -3,11 +3,17 @@ export {
|
|
|
3
3
|
chat,
|
|
4
4
|
summarize,
|
|
5
5
|
rerank,
|
|
6
|
+
decide,
|
|
7
|
+
choice,
|
|
8
|
+
score,
|
|
9
|
+
boolean,
|
|
6
10
|
generateImage,
|
|
7
11
|
generateAudio,
|
|
8
12
|
generateVideo,
|
|
9
13
|
getVideoJobStatus,
|
|
10
14
|
generateSpeech,
|
|
15
|
+
listVoices,
|
|
16
|
+
generateVoice,
|
|
11
17
|
generateTranscription,
|
|
12
18
|
embed,
|
|
13
19
|
generateWorld,
|
|
@@ -22,6 +28,7 @@ export { createImageOptions } from './activities/generateImage/index'
|
|
|
22
28
|
export { createAudioOptions } from './activities/generateAudio/index'
|
|
23
29
|
export { createVideoOptions } from './activities/generateVideo/index'
|
|
24
30
|
export { createSpeechOptions } from './activities/generateSpeech/index'
|
|
31
|
+
export { createVoiceOptions } from './activities/generateVoice/index'
|
|
25
32
|
export { createTranscriptionOptions } from './activities/generateTranscription/index'
|
|
26
33
|
export { createEmbedOptions } from './activities/embed/index'
|
|
27
34
|
export { createWorldOptions } from './activities/generateWorld/index'
|
|
@@ -40,6 +47,9 @@ export type {
|
|
|
40
47
|
AudioAdapter,
|
|
41
48
|
AnyTTSAdapter,
|
|
42
49
|
TTSAdapter,
|
|
50
|
+
TTSCapabilities,
|
|
51
|
+
AnyVoiceAdapter,
|
|
52
|
+
VoiceAdapter,
|
|
43
53
|
AnyTranscriptionAdapter,
|
|
44
54
|
TranscriptionAdapter,
|
|
45
55
|
AnyVideoAdapter,
|
|
@@ -48,6 +58,14 @@ export type {
|
|
|
48
58
|
EmbeddingAdapter,
|
|
49
59
|
AnyRerankAdapter,
|
|
50
60
|
RerankAdapter,
|
|
61
|
+
AnyEvaluateAdapter,
|
|
62
|
+
EvaluateAdapter,
|
|
63
|
+
ChoiceAnswer,
|
|
64
|
+
ScoreAnswer,
|
|
65
|
+
BooleanAnswer,
|
|
66
|
+
EvaluateResult,
|
|
67
|
+
WireQuestion,
|
|
68
|
+
WireAnswer,
|
|
51
69
|
AnyWorldAdapter,
|
|
52
70
|
WorldAdapter,
|
|
53
71
|
AnyLiveVideoAdapter,
|
|
@@ -57,6 +75,9 @@ export type {
|
|
|
57
75
|
// Rerank adapter base + types
|
|
58
76
|
export { BaseRerankAdapter } from './activities/rerank/adapter'
|
|
59
77
|
|
|
78
|
+
// Evaluate adapter base + types
|
|
79
|
+
export { BaseEvaluateAdapter } from './activities/evaluate/adapter'
|
|
80
|
+
|
|
60
81
|
// Tool definition
|
|
61
82
|
export {
|
|
62
83
|
toolDefinition,
|
package/src/middlewares/otel.ts
CHANGED
|
@@ -87,9 +87,11 @@ const OPERATION_NAME: Record<GenerationActivity, string> = {
|
|
|
87
87
|
video: 'video_generation',
|
|
88
88
|
audio: 'audio_generation',
|
|
89
89
|
tts: 'text_to_speech',
|
|
90
|
+
voice: 'voice_generation',
|
|
90
91
|
transcription: 'transcription',
|
|
91
92
|
embedding: 'embeddings',
|
|
92
93
|
rerank: 'rerank',
|
|
94
|
+
evaluate: 'evaluate',
|
|
93
95
|
summarize: 'summarize',
|
|
94
96
|
world: 'world_generation',
|
|
95
97
|
liveVideo: 'live_video_generation',
|
package/src/realtime/index.ts
CHANGED
|
@@ -22,7 +22,7 @@ export type * from './types'
|
|
|
22
22
|
* // On the server (e.g. inside a server route or framework server
|
|
23
23
|
* // function), mint an ephemeral token for the client:
|
|
24
24
|
* const token = await realtimeToken({
|
|
25
|
-
* adapter: openaiRealtimeToken({ model: 'gpt-realtime' }),
|
|
25
|
+
* adapter: openaiRealtimeToken({ model: 'gpt-realtime-2.1' }),
|
|
26
26
|
* })
|
|
27
27
|
* ```
|
|
28
28
|
*/
|
package/src/types.ts
CHANGED
|
@@ -2350,6 +2350,56 @@ export interface LiveVideoGenerationResult {
|
|
|
2350
2350
|
// Text-to-Speech (TTS) Types
|
|
2351
2351
|
// ============================================================================
|
|
2352
2352
|
|
|
2353
|
+
/**
|
|
2354
|
+
* One turn of a multi-voice dialogue request.
|
|
2355
|
+
*
|
|
2356
|
+
* Providers that expose a dedicated dialogue endpoint (ElevenLabs
|
|
2357
|
+
* `textToDialogue`, Gemini multi-speaker) take these natively instead of a
|
|
2358
|
+
* single `text` + `voice` pair.
|
|
2359
|
+
*/
|
|
2360
|
+
export interface TTSTurn {
|
|
2361
|
+
/** The text this voice speaks. */
|
|
2362
|
+
text: string
|
|
2363
|
+
/** Provider voice id (ElevenLabs) or voice name (Gemini) for this turn. */
|
|
2364
|
+
voice: string
|
|
2365
|
+
}
|
|
2366
|
+
|
|
2367
|
+
/**
|
|
2368
|
+
* Timings for the generated audio, returned when `timestamps: true` was
|
|
2369
|
+
* requested and the adapter declares `capabilities.timestamps`.
|
|
2370
|
+
*
|
|
2371
|
+
* Granularity differs per provider — ElevenLabs reports characters, BytePlus
|
|
2372
|
+
* reports words — so `unit` says which, and the three arrays are parallel.
|
|
2373
|
+
* All times are **seconds**; adapters convert.
|
|
2374
|
+
*/
|
|
2375
|
+
export interface TTSAlignment {
|
|
2376
|
+
/** Granularity of each entry. */
|
|
2377
|
+
unit: 'character' | 'word'
|
|
2378
|
+
/** Entry text, in audio order. */
|
|
2379
|
+
texts: Array<string>
|
|
2380
|
+
/** Start of each entry in seconds. Same length as `texts`. */
|
|
2381
|
+
startSeconds: Array<number>
|
|
2382
|
+
/** End of each entry in seconds. Same length as `texts`. */
|
|
2383
|
+
endSeconds: Array<number>
|
|
2384
|
+
}
|
|
2385
|
+
|
|
2386
|
+
/**
|
|
2387
|
+
* A stretch of audio attributable to one turn (multi-voice) or one utterance
|
|
2388
|
+
* (single voice). This is what tells a consumer which turn is where.
|
|
2389
|
+
*/
|
|
2390
|
+
export interface TTSSegment {
|
|
2391
|
+
/** Start of the segment in seconds. */
|
|
2392
|
+
startSeconds: number
|
|
2393
|
+
/** End of the segment in seconds. */
|
|
2394
|
+
endSeconds: number
|
|
2395
|
+
/** Index into the request's `turns`, when the provider reports it. */
|
|
2396
|
+
turnIndex?: number
|
|
2397
|
+
/** Voice heard in this segment, when the provider reports it. */
|
|
2398
|
+
voice?: string
|
|
2399
|
+
/** Text spoken in this segment, when the provider reports it. */
|
|
2400
|
+
text?: string
|
|
2401
|
+
}
|
|
2402
|
+
|
|
2353
2403
|
/**
|
|
2354
2404
|
* Options for text-to-speech generation.
|
|
2355
2405
|
* These are the common options supported across providers.
|
|
@@ -2357,8 +2407,24 @@ export interface LiveVideoGenerationResult {
|
|
|
2357
2407
|
export interface TTSOptions<TProviderOptions extends object = object> {
|
|
2358
2408
|
/** The model to use for TTS generation */
|
|
2359
2409
|
model: string
|
|
2360
|
-
/**
|
|
2410
|
+
/**
|
|
2411
|
+
* The text to convert to speech. When the caller passed `turns`, the
|
|
2412
|
+
* activity fills this with the turn texts joined by newlines so adapters
|
|
2413
|
+
* that only read `text` still receive the full script.
|
|
2414
|
+
*/
|
|
2361
2415
|
text: string
|
|
2416
|
+
/**
|
|
2417
|
+
* Multi-voice dialogue turns, when the caller asked for dialogue. Only
|
|
2418
|
+
* adapters that declare `capabilities.maxSpeakers` ever see this — the
|
|
2419
|
+
* activity rejects `turns` for the rest.
|
|
2420
|
+
*/
|
|
2421
|
+
turns?: Array<TTSTurn>
|
|
2422
|
+
/**
|
|
2423
|
+
* Ask for `alignment` / `segments` on the result. Rejected by the activity
|
|
2424
|
+
* unless the adapter declares `capabilities.timestamps`, because on some
|
|
2425
|
+
* providers this is a different endpoint rather than free metadata.
|
|
2426
|
+
*/
|
|
2427
|
+
timestamps?: boolean
|
|
2362
2428
|
/** The voice to use for generation */
|
|
2363
2429
|
voice?: string
|
|
2364
2430
|
/** The output audio format */
|
|
@@ -2393,8 +2459,18 @@ export interface TTSResult {
|
|
|
2393
2459
|
audio: string
|
|
2394
2460
|
/** Audio format of the generated audio */
|
|
2395
2461
|
format: string
|
|
2396
|
-
/** Duration of the audio in seconds, if available */
|
|
2462
|
+
/** Duration of the audio file in seconds, if available */
|
|
2397
2463
|
duration?: number
|
|
2464
|
+
/**
|
|
2465
|
+
* Character- or word-level timings, present when `timestamps: true` was
|
|
2466
|
+
* requested. Use this rather than `duration` to find where *speech* ends.
|
|
2467
|
+
*/
|
|
2468
|
+
alignment?: TTSAlignment
|
|
2469
|
+
/**
|
|
2470
|
+
* Per-turn (or per-utterance) spans of the audio, present when
|
|
2471
|
+
* `timestamps: true` was requested and the provider reports segmentation.
|
|
2472
|
+
*/
|
|
2473
|
+
segments?: Array<TTSSegment>
|
|
2398
2474
|
/** Content type of the audio (e.g., 'audio/mp3') */
|
|
2399
2475
|
contentType?: string
|
|
2400
2476
|
/** Token usage information (if provided by the adapter) */
|
|
@@ -2403,6 +2479,174 @@ export interface TTSResult {
|
|
|
2403
2479
|
artifacts?: Array<PersistedArtifactRef>
|
|
2404
2480
|
}
|
|
2405
2481
|
|
|
2482
|
+
// ============================================================================
|
|
2483
|
+
// Voice Catalog Types
|
|
2484
|
+
// ============================================================================
|
|
2485
|
+
|
|
2486
|
+
/**
|
|
2487
|
+
* Where a voice in a provider's catalog came from.
|
|
2488
|
+
*
|
|
2489
|
+
* `'premade'` is the provider's own stock catalog. `'generated'` and
|
|
2490
|
+
* `'cloned'` are voices the account made, which is what `generateVoice()`
|
|
2491
|
+
* produces. `'professional'` covers a provider's curated or paid tiers.
|
|
2492
|
+
*/
|
|
2493
|
+
export type VoiceOrigin = 'premade' | 'generated' | 'cloned' | 'professional'
|
|
2494
|
+
|
|
2495
|
+
/** One voice from a provider's catalog. */
|
|
2496
|
+
export interface CatalogVoice {
|
|
2497
|
+
/** Pass this to `generateSpeech()` as `voice` */
|
|
2498
|
+
voiceId: string
|
|
2499
|
+
/** Display name, when the provider stores one */
|
|
2500
|
+
name?: string
|
|
2501
|
+
/** Where the voice came from */
|
|
2502
|
+
origin?: VoiceOrigin
|
|
2503
|
+
/** Provider description of the voice */
|
|
2504
|
+
description?: string
|
|
2505
|
+
/** URL of a sample, when the provider hosts one */
|
|
2506
|
+
previewUrl?: string
|
|
2507
|
+
/** Provider labels, such as accent, age, or use case */
|
|
2508
|
+
labels?: Record<string, string>
|
|
2509
|
+
}
|
|
2510
|
+
|
|
2511
|
+
/** Options for listing a provider's voices. */
|
|
2512
|
+
export interface ListVoicesOptions {
|
|
2513
|
+
/**
|
|
2514
|
+
* Restrict the result to voices of these origins. Adapters filter server
|
|
2515
|
+
* side when the provider supports it, and in memory otherwise.
|
|
2516
|
+
*/
|
|
2517
|
+
origins?: Array<VoiceOrigin>
|
|
2518
|
+
/**
|
|
2519
|
+
* Effective abort signal. Adapters forward this to the provider SDK when
|
|
2520
|
+
* supported.
|
|
2521
|
+
*/
|
|
2522
|
+
abortSignal?: AbortSignal
|
|
2523
|
+
}
|
|
2524
|
+
|
|
2525
|
+
/** Result of listing a provider's voices. */
|
|
2526
|
+
export interface ListVoicesResult {
|
|
2527
|
+
/** The voices available to this account */
|
|
2528
|
+
voices: Array<CatalogVoice>
|
|
2529
|
+
}
|
|
2530
|
+
|
|
2531
|
+
// ============================================================================
|
|
2532
|
+
// Voice Creation Types
|
|
2533
|
+
// ============================================================================
|
|
2534
|
+
|
|
2535
|
+
/**
|
|
2536
|
+
* Options for creating a voice.
|
|
2537
|
+
*
|
|
2538
|
+
* Providers create voices in one of two ways, and some support both:
|
|
2539
|
+
* - **design** — synthesize a brand new voice from a text {@link prompt}.
|
|
2540
|
+
* - **clone** — derive a voice from {@link referenceAudio} of a real speaker.
|
|
2541
|
+
*
|
|
2542
|
+
* At least one of `prompt` / `referenceAudio` is required; which ones an
|
|
2543
|
+
* adapter accepts depends on the model. An adapter may require both — the
|
|
2544
|
+
* only adapter today, `elevenlabsVoiceDesign`, always needs `prompt` and
|
|
2545
|
+
* takes `referenceAudio` as an additional design reference.
|
|
2546
|
+
*/
|
|
2547
|
+
export interface VoiceGenerationOptions<
|
|
2548
|
+
TProviderOptions extends object = object,
|
|
2549
|
+
> {
|
|
2550
|
+
/** The model to use for voice creation */
|
|
2551
|
+
model: string
|
|
2552
|
+
/** Text description of the voice to create, for design-capable models */
|
|
2553
|
+
prompt?: string
|
|
2554
|
+
/**
|
|
2555
|
+
* Reference audio of the speaker to clone - base64 string, base64 data URL,
|
|
2556
|
+
* File, Blob, or ArrayBuffer. For clone-capable models. Remote URLs are not
|
|
2557
|
+
* accepted; read the file and pass the bytes.
|
|
2558
|
+
*/
|
|
2559
|
+
referenceAudio?: string | File | Blob | ArrayBuffer
|
|
2560
|
+
/**
|
|
2561
|
+
* Name to store the voice under in the provider's voice library. Providers
|
|
2562
|
+
* differ on what this implies — ElevenLabs only persists a designed voice
|
|
2563
|
+
* when a name is given. Read {@link GeneratedVoice.saved} to find out what
|
|
2564
|
+
* actually happened.
|
|
2565
|
+
*/
|
|
2566
|
+
name?: string
|
|
2567
|
+
/** Human-readable description stored alongside the voice */
|
|
2568
|
+
description?: string
|
|
2569
|
+
/** Model-specific options for voice creation */
|
|
2570
|
+
modelOptions?: TProviderOptions
|
|
2571
|
+
/**
|
|
2572
|
+
* Internal logger threaded from the generateVoice() entry point. Adapters
|
|
2573
|
+
* must call logger.request() before the SDK call and logger.errors() in
|
|
2574
|
+
* catch blocks.
|
|
2575
|
+
*/
|
|
2576
|
+
logger: InternalLogger
|
|
2577
|
+
/**
|
|
2578
|
+
* Effective abort signal composed by the activity from caller `abortSignal`
|
|
2579
|
+
* and/or `timeout`. Adapters should forward this to the provider SDK when
|
|
2580
|
+
* supported. Request-specific - never store on a global client config.
|
|
2581
|
+
*/
|
|
2582
|
+
abortSignal?: AbortSignal
|
|
2583
|
+
}
|
|
2584
|
+
|
|
2585
|
+
/**
|
|
2586
|
+
* A single voice produced by {@link VoiceGenerationOptions}.
|
|
2587
|
+
*/
|
|
2588
|
+
export interface GeneratedVoice {
|
|
2589
|
+
/**
|
|
2590
|
+
* The provider's voice identifier. Pass it straight back as the `voice`
|
|
2591
|
+
* option on `generateSpeech()`.
|
|
2592
|
+
*/
|
|
2593
|
+
voiceId: string
|
|
2594
|
+
/** Base64-encoded preview audio, when the provider returns one */
|
|
2595
|
+
audio?: string
|
|
2596
|
+
/** Audio format of the preview (e.g. 'mp3') */
|
|
2597
|
+
format?: string
|
|
2598
|
+
/** Content type of the preview (e.g. 'audio/mpeg') */
|
|
2599
|
+
contentType?: string
|
|
2600
|
+
/** Duration of the preview in seconds, if available */
|
|
2601
|
+
duration?: number
|
|
2602
|
+
/** Language of the preview, if reported */
|
|
2603
|
+
language?: string
|
|
2604
|
+
/**
|
|
2605
|
+
* Whether the voice is persisted in the provider's voice library. Unsaved
|
|
2606
|
+
* voices are previews and generally expire.
|
|
2607
|
+
*/
|
|
2608
|
+
saved: boolean
|
|
2609
|
+
/**
|
|
2610
|
+
* Whether the voice can be used in `generateSpeech()` yet. Required so a
|
|
2611
|
+
* caller never has to guess: every adapter states it outright.
|
|
2612
|
+
*/
|
|
2613
|
+
status: VoiceTrainingStatus
|
|
2614
|
+
}
|
|
2615
|
+
|
|
2616
|
+
/**
|
|
2617
|
+
* Whether a created voice is usable.
|
|
2618
|
+
*
|
|
2619
|
+
* - `'ready'` — usable in `generateSpeech()` now. Every adapter today returns
|
|
2620
|
+
* this, because they all finish the voice inside `generateVoice()`.
|
|
2621
|
+
* - `'training'` — the provider accepted the request but is still building
|
|
2622
|
+
* the voice, so it is not usable yet. Reserved for providers that train
|
|
2623
|
+
* asynchronously; no adapter returns it yet, and reading the state back
|
|
2624
|
+
* will land with the first adapter that needs it.
|
|
2625
|
+
* - `'failed'` — the provider finished without producing a usable voice.
|
|
2626
|
+
*/
|
|
2627
|
+
export type VoiceTrainingStatus = 'ready' | 'training' | 'failed'
|
|
2628
|
+
|
|
2629
|
+
/**
|
|
2630
|
+
* Result of voice creation.
|
|
2631
|
+
*
|
|
2632
|
+
* Design models typically return several candidates to choose between; clone
|
|
2633
|
+
* models return exactly one.
|
|
2634
|
+
*/
|
|
2635
|
+
export interface VoiceResult {
|
|
2636
|
+
/** Unique identifier for the generation */
|
|
2637
|
+
id: string
|
|
2638
|
+
/** Model used for generation */
|
|
2639
|
+
model: string
|
|
2640
|
+
/** The voices produced, best-first when the provider ranks them */
|
|
2641
|
+
voices: Array<GeneratedVoice>
|
|
2642
|
+
/** The line spoken in the previews, when the provider generated one */
|
|
2643
|
+
previewText?: string
|
|
2644
|
+
/** Token usage information (if provided by the adapter) */
|
|
2645
|
+
usage?: TokenUsage
|
|
2646
|
+
/** Persisted artifact references for generated assets, when available */
|
|
2647
|
+
artifacts?: Array<PersistedArtifactRef>
|
|
2648
|
+
}
|
|
2649
|
+
|
|
2406
2650
|
// ============================================================================
|
|
2407
2651
|
// Transcription (Speech-to-Text) Types
|
|
2408
2652
|
// ============================================================================
|