@tanstack/ai 0.55.0 → 0.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +42 -16
  2. package/dist/esm/activities/chat/tools/tool-calls.js +1 -0
  3. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  4. package/dist/esm/activities/evaluate/adapter.d.ts +160 -0
  5. package/dist/esm/activities/evaluate/adapter.js +23 -0
  6. package/dist/esm/activities/evaluate/adapter.js.map +1 -0
  7. package/dist/esm/activities/evaluate/index.d.ts +255 -0
  8. package/dist/esm/activities/evaluate/index.js +317 -0
  9. package/dist/esm/activities/evaluate/index.js.map +1 -0
  10. package/dist/esm/activities/generateSpeech/adapter.d.ts +39 -1
  11. package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
  12. package/dist/esm/activities/generateSpeech/index.d.ts +55 -5
  13. package/dist/esm/activities/generateSpeech/index.js +53 -3
  14. package/dist/esm/activities/generateSpeech/index.js.map +1 -1
  15. package/dist/esm/activities/generateVoice/adapter.d.ts +62 -0
  16. package/dist/esm/activities/generateVoice/adapter.js +23 -0
  17. package/dist/esm/activities/generateVoice/adapter.js.map +1 -0
  18. package/dist/esm/activities/generateVoice/index.d.ts +133 -0
  19. package/dist/esm/activities/generateVoice/index.js +184 -0
  20. package/dist/esm/activities/generateVoice/index.js.map +1 -0
  21. package/dist/esm/activities/index.d.ts +10 -4
  22. package/dist/esm/activities/index.js +14 -10
  23. package/dist/esm/activities/middleware/types.d.ts +1 -1
  24. package/dist/esm/client.d.ts +3 -2
  25. package/dist/esm/client.js +21 -3
  26. package/dist/esm/client.js.map +1 -1
  27. package/dist/esm/index.d.ts +4 -2
  28. package/dist/esm/index.js +5 -2
  29. package/dist/esm/middlewares/otel.js +2 -0
  30. package/dist/esm/middlewares/otel.js.map +1 -1
  31. package/dist/esm/realtime/index.d.ts +1 -1
  32. package/dist/esm/realtime/index.js +1 -1
  33. package/dist/esm/realtime/index.js.map +1 -1
  34. package/dist/esm/types.d.ts +225 -2
  35. package/package.json +3 -3
  36. package/skills/ai-core/media-generation/SKILL.md +132 -6
  37. package/src/activities/chat/tools/tool-calls.ts +9 -0
  38. package/src/activities/evaluate/adapter.ts +212 -0
  39. package/src/activities/evaluate/index.ts +614 -0
  40. package/src/activities/generateSpeech/adapter.ts +47 -1
  41. package/src/activities/generateSpeech/index.ts +149 -8
  42. package/src/activities/generateVoice/adapter.ts +89 -0
  43. package/src/activities/generateVoice/index.ts +371 -0
  44. package/src/activities/index.ts +69 -0
  45. package/src/activities/middleware/types.ts +2 -0
  46. package/src/client.ts +35 -8
  47. package/src/index.ts +21 -0
  48. package/src/middlewares/otel.ts +2 -0
  49. package/src/realtime/index.ts +1 -1
  50. package/src/types.ts +246 -2
@@ -20,9 +20,11 @@ import type { AnyImageAdapter } from './generateImage/adapter'
20
20
  import type { AnyAudioAdapter } from './generateAudio/adapter'
21
21
  import type { AnyVideoAdapter } from './generateVideo/adapter'
22
22
  import type { AnyTTSAdapter } from './generateSpeech/adapter'
23
+ import type { AnyVoiceAdapter } from './generateVoice/adapter'
23
24
  import type { AnyTranscriptionAdapter } from './generateTranscription/adapter'
24
25
  import type { AnyEmbeddingAdapter } from './embed/adapter'
25
26
  import type { AnyRerankAdapter } from './rerank/adapter'
27
+ import type { AnyEvaluateAdapter } from './evaluate/adapter'
26
28
  import type { AnyWorldAdapter } from './generateWorld/adapter'
27
29
  import type { AnyLiveVideoAdapter } from './generateLiveVideo/adapter'
28
30
 
@@ -89,6 +91,46 @@ export {
89
91
  type AnyRerankAdapter,
90
92
  } from './rerank/adapter'
91
93
 
94
+ // ===========================
95
+ // Evaluate Activity
96
+ // ===========================
97
+
98
+ export {
99
+ kind as evaluateKind,
100
+ decide,
101
+ choice,
102
+ score,
103
+ boolean,
104
+ type EvaluateActivityOptions,
105
+ type EvaluateResult,
106
+ type EvaluateResultMeta,
107
+ type EvaluateProviderOptions,
108
+ type ChoiceAnswer,
109
+ type ScoreAnswer,
110
+ type BooleanAnswer,
111
+ type InferEvaluateAnswer,
112
+ } from './evaluate/index'
113
+
114
+ export {
115
+ BaseEvaluateAdapter,
116
+ type EvaluateAdapter,
117
+ type EvaluateAdapterConfig,
118
+ type AnyEvaluateAdapter,
119
+ type EvaluateOptions,
120
+ type EvaluateAdapterResult,
121
+ type EvaluateState,
122
+ type EvaluateInstructions,
123
+ type EvaluateJsonValue,
124
+ type WireQuestion,
125
+ type WireAnswer,
126
+ type WireChoiceQuestion,
127
+ type WireScoreQuestion,
128
+ type WireNoulQuestion,
129
+ type WireChoiceAnswer,
130
+ type WireScoreAnswer,
131
+ type WireNoulAnswer,
132
+ } from './evaluate/adapter'
133
+
92
134
  // ===========================
93
135
  // Image Activity
94
136
  // ===========================
@@ -162,6 +204,8 @@ export { snapToDurationOption } from './generateVideo/snap'
162
204
  export {
163
205
  kind as ttsKind,
164
206
  generateSpeech,
207
+ listVoices,
208
+ type ListVoicesActivityOptions,
165
209
  type TTSActivityOptions,
166
210
  type TTSActivityResult,
167
211
  type TTSProviderOptions,
@@ -171,9 +215,30 @@ export {
171
215
  BaseTTSAdapter,
172
216
  type TTSAdapter,
173
217
  type TTSAdapterConfig,
218
+ type TTSCapabilities,
174
219
  type AnyTTSAdapter,
175
220
  } from './generateSpeech/adapter'
176
221
 
222
+ // ===========================
223
+ // Voice Activity
224
+ // ===========================
225
+
226
+ export {
227
+ kind as voiceKind,
228
+ generateVoice,
229
+ createVoiceOptions,
230
+ type VoiceActivityOptions,
231
+ type VoiceActivityResult,
232
+ type VoiceProviderOptions,
233
+ } from './generateVoice/index'
234
+
235
+ export {
236
+ BaseVoiceAdapter,
237
+ type VoiceAdapter,
238
+ type VoiceAdapterConfig,
239
+ type AnyVoiceAdapter,
240
+ } from './generateVoice/adapter'
241
+
177
242
  // ===========================
178
243
  // Transcription Activity
179
244
  // ===========================
@@ -262,9 +327,11 @@ export type AIAdapter =
262
327
  | AnyAudioAdapter
263
328
  | AnyVideoAdapter
264
329
  | AnyTTSAdapter
330
+ | AnyVoiceAdapter
265
331
  | AnyTranscriptionAdapter
266
332
  | AnyEmbeddingAdapter
267
333
  | AnyRerankAdapter
334
+ | AnyEvaluateAdapter
268
335
  | AnyWorldAdapter
269
336
  | AnyLiveVideoAdapter
270
337
 
@@ -276,8 +343,10 @@ export type AdapterKind =
276
343
  | 'audio'
277
344
  | 'video'
278
345
  | 'tts'
346
+ | 'voice'
279
347
  | 'transcription'
280
348
  | 'embedding'
281
349
  | 'rerank'
350
+ | 'evaluate'
282
351
  | 'world'
283
352
  | 'liveVideo'
@@ -40,9 +40,11 @@ export type GenerationActivity =
40
40
  | 'video'
41
41
  | 'audio'
42
42
  | 'tts'
43
+ | 'voice'
43
44
  | 'transcription'
44
45
  | 'embedding'
45
46
  | 'rerank'
47
+ | 'evaluate'
46
48
  | 'summarize'
47
49
  | 'world'
48
50
  | 'liveVideo'
package/src/client.ts CHANGED
@@ -3,6 +3,7 @@ import type {
3
3
  ImageGenerationOptions,
4
4
  TTSOptions,
5
5
  TranscriptionOptions,
6
+ VoiceGenerationOptions,
6
7
  VideoGenerationOptions,
7
8
  WorldGenerationOptions,
8
9
  LiveVideoGenerationOptions,
@@ -12,6 +13,7 @@ export type GenerationKind =
12
13
  | 'image'
13
14
  | 'audio'
14
15
  | 'tts'
16
+ | 'voice'
15
17
  | 'video'
16
18
  | 'transcription'
17
19
  | 'world'
@@ -21,6 +23,7 @@ type GenerationInputByKind = {
21
23
  image: Omit<ImageGenerationOptions, 'logger' | 'model'>
22
24
  audio: Omit<AudioGenerationOptions, 'logger' | 'model'>
23
25
  tts: Omit<TTSOptions, 'logger' | 'model'>
26
+ voice: Omit<VoiceGenerationOptions, 'logger' | 'model'>
24
27
  video: Omit<VideoGenerationOptions, 'logger' | 'model'>
25
28
  transcription: Omit<TranscriptionOptions, 'logger' | 'model'>
26
29
  world: Omit<WorldGenerationOptions, 'logger' | 'model'>
@@ -38,6 +41,7 @@ const generationKinds = [
38
41
  'image',
39
42
  'audio',
40
43
  'tts',
44
+ 'voice',
41
45
  'video',
42
46
  'transcription',
43
47
  'world',
@@ -69,6 +73,31 @@ function assertGenerationKind(kind: unknown): asserts kind is GenerationKind {
69
73
  }
70
74
  }
71
75
 
76
+ /**
77
+ * The input field(s) that identify a generation body for a kind. Most kinds
78
+ * have exactly one; `voice` accepts either of its two creation modes, so any
79
+ * one of its keys is enough.
80
+ */
81
+ function requiredKeysForKind(kind: GenerationKind): Array<string> {
82
+ // Enumerated rather than defaulted so a new generation kind has to declare
83
+ // the field that identifies its body instead of silently inheriting
84
+ // `prompt`.
85
+ switch (kind) {
86
+ case 'tts':
87
+ return ['text']
88
+ case 'transcription':
89
+ return ['audio']
90
+ case 'voice':
91
+ return ['prompt', 'referenceAudio']
92
+ case 'image':
93
+ case 'audio':
94
+ case 'video':
95
+ case 'world':
96
+ case 'liveVideo':
97
+ return ['prompt']
98
+ }
99
+ }
100
+
72
101
  function assertInputForKind(
73
102
  kind: GenerationKind,
74
103
  input: unknown,
@@ -77,21 +106,19 @@ function assertInputForKind(
77
106
  throw new Error(`Generation ${kind} input must be an object.`)
78
107
  }
79
108
 
80
- const requiredKey =
81
- kind === 'tts' ? 'text' : kind === 'transcription' ? 'audio' : 'prompt'
109
+ const requiredKeys = requiredKeysForKind(kind)
82
110
 
83
- if (!hasOwnKey(input, requiredKey)) {
84
- throw new Error(`Generation ${kind} input must include ${requiredKey}.`)
111
+ if (!requiredKeys.some((key) => hasOwnKey(input, key))) {
112
+ throw new Error(
113
+ `Generation ${kind} input must include ${requiredKeys.join(' or ')}.`,
114
+ )
85
115
  }
86
116
  }
87
117
 
88
118
  function isInputForKind(kind: GenerationKind, input: unknown): boolean {
89
119
  if (!isRecord(input)) return false
90
120
 
91
- const requiredKey =
92
- kind === 'tts' ? 'text' : kind === 'transcription' ? 'audio' : 'prompt'
93
-
94
- return hasOwnKey(input, requiredKey)
121
+ return requiredKeysForKind(kind).some((key) => hasOwnKey(input, key))
95
122
  }
96
123
 
97
124
  function forwardedPropsFromEnvelope(
package/src/index.ts CHANGED
@@ -3,11 +3,17 @@ export {
3
3
  chat,
4
4
  summarize,
5
5
  rerank,
6
+ decide,
7
+ choice,
8
+ score,
9
+ boolean,
6
10
  generateImage,
7
11
  generateAudio,
8
12
  generateVideo,
9
13
  getVideoJobStatus,
10
14
  generateSpeech,
15
+ listVoices,
16
+ generateVoice,
11
17
  generateTranscription,
12
18
  embed,
13
19
  generateWorld,
@@ -22,6 +28,7 @@ export { createImageOptions } from './activities/generateImage/index'
22
28
  export { createAudioOptions } from './activities/generateAudio/index'
23
29
  export { createVideoOptions } from './activities/generateVideo/index'
24
30
  export { createSpeechOptions } from './activities/generateSpeech/index'
31
+ export { createVoiceOptions } from './activities/generateVoice/index'
25
32
  export { createTranscriptionOptions } from './activities/generateTranscription/index'
26
33
  export { createEmbedOptions } from './activities/embed/index'
27
34
  export { createWorldOptions } from './activities/generateWorld/index'
@@ -40,6 +47,9 @@ export type {
40
47
  AudioAdapter,
41
48
  AnyTTSAdapter,
42
49
  TTSAdapter,
50
+ TTSCapabilities,
51
+ AnyVoiceAdapter,
52
+ VoiceAdapter,
43
53
  AnyTranscriptionAdapter,
44
54
  TranscriptionAdapter,
45
55
  AnyVideoAdapter,
@@ -48,6 +58,14 @@ export type {
48
58
  EmbeddingAdapter,
49
59
  AnyRerankAdapter,
50
60
  RerankAdapter,
61
+ AnyEvaluateAdapter,
62
+ EvaluateAdapter,
63
+ ChoiceAnswer,
64
+ ScoreAnswer,
65
+ BooleanAnswer,
66
+ EvaluateResult,
67
+ WireQuestion,
68
+ WireAnswer,
51
69
  AnyWorldAdapter,
52
70
  WorldAdapter,
53
71
  AnyLiveVideoAdapter,
@@ -57,6 +75,9 @@ export type {
57
75
  // Rerank adapter base + types
58
76
  export { BaseRerankAdapter } from './activities/rerank/adapter'
59
77
 
78
+ // Evaluate adapter base + types
79
+ export { BaseEvaluateAdapter } from './activities/evaluate/adapter'
80
+
60
81
  // Tool definition
61
82
  export {
62
83
  toolDefinition,
@@ -87,9 +87,11 @@ const OPERATION_NAME: Record<GenerationActivity, string> = {
87
87
  video: 'video_generation',
88
88
  audio: 'audio_generation',
89
89
  tts: 'text_to_speech',
90
+ voice: 'voice_generation',
90
91
  transcription: 'transcription',
91
92
  embedding: 'embeddings',
92
93
  rerank: 'rerank',
94
+ evaluate: 'evaluate',
93
95
  summarize: 'summarize',
94
96
  world: 'world_generation',
95
97
  liveVideo: 'live_video_generation',
@@ -22,7 +22,7 @@ export type * from './types'
22
22
  * // On the server (e.g. inside a server route or framework server
23
23
  * // function), mint an ephemeral token for the client:
24
24
  * const token = await realtimeToken({
25
- * adapter: openaiRealtimeToken({ model: 'gpt-realtime' }),
25
+ * adapter: openaiRealtimeToken({ model: 'gpt-realtime-2.1' }),
26
26
  * })
27
27
  * ```
28
28
  */
package/src/types.ts CHANGED
@@ -2350,6 +2350,56 @@ export interface LiveVideoGenerationResult {
2350
2350
  // Text-to-Speech (TTS) Types
2351
2351
  // ============================================================================
2352
2352
 
2353
+ /**
2354
+ * One turn of a multi-voice dialogue request.
2355
+ *
2356
+ * Providers that expose a dedicated dialogue endpoint (ElevenLabs
2357
+ * `textToDialogue`, Gemini multi-speaker) take these natively instead of a
2358
+ * single `text` + `voice` pair.
2359
+ */
2360
+ export interface TTSTurn {
2361
+ /** The text this voice speaks. */
2362
+ text: string
2363
+ /** Provider voice id (ElevenLabs) or voice name (Gemini) for this turn. */
2364
+ voice: string
2365
+ }
2366
+
2367
+ /**
2368
+ * Timings for the generated audio, returned when `timestamps: true` was
2369
+ * requested and the adapter declares `capabilities.timestamps`.
2370
+ *
2371
+ * Granularity differs per provider — ElevenLabs reports characters, BytePlus
2372
+ * reports words — so `unit` says which, and the three arrays are parallel.
2373
+ * All times are **seconds**; adapters convert.
2374
+ */
2375
+ export interface TTSAlignment {
2376
+ /** Granularity of each entry. */
2377
+ unit: 'character' | 'word'
2378
+ /** Entry text, in audio order. */
2379
+ texts: Array<string>
2380
+ /** Start of each entry in seconds. Same length as `texts`. */
2381
+ startSeconds: Array<number>
2382
+ /** End of each entry in seconds. Same length as `texts`. */
2383
+ endSeconds: Array<number>
2384
+ }
2385
+
2386
+ /**
2387
+ * A stretch of audio attributable to one turn (multi-voice) or one utterance
2388
+ * (single voice). This is what tells a consumer which turn is where.
2389
+ */
2390
+ export interface TTSSegment {
2391
+ /** Start of the segment in seconds. */
2392
+ startSeconds: number
2393
+ /** End of the segment in seconds. */
2394
+ endSeconds: number
2395
+ /** Index into the request's `turns`, when the provider reports it. */
2396
+ turnIndex?: number
2397
+ /** Voice heard in this segment, when the provider reports it. */
2398
+ voice?: string
2399
+ /** Text spoken in this segment, when the provider reports it. */
2400
+ text?: string
2401
+ }
2402
+
2353
2403
  /**
2354
2404
  * Options for text-to-speech generation.
2355
2405
  * These are the common options supported across providers.
@@ -2357,8 +2407,24 @@ export interface LiveVideoGenerationResult {
2357
2407
  export interface TTSOptions<TProviderOptions extends object = object> {
2358
2408
  /** The model to use for TTS generation */
2359
2409
  model: string
2360
- /** The text to convert to speech */
2410
+ /**
2411
+ * The text to convert to speech. When the caller passed `turns`, the
2412
+ * activity fills this with the turn texts joined by newlines so adapters
2413
+ * that only read `text` still receive the full script.
2414
+ */
2361
2415
  text: string
2416
+ /**
2417
+ * Multi-voice dialogue turns, when the caller asked for dialogue. Only
2418
+ * adapters that declare `capabilities.maxSpeakers` ever see this — the
2419
+ * activity rejects `turns` for the rest.
2420
+ */
2421
+ turns?: Array<TTSTurn>
2422
+ /**
2423
+ * Ask for `alignment` / `segments` on the result. Rejected by the activity
2424
+ * unless the adapter declares `capabilities.timestamps`, because on some
2425
+ * providers this is a different endpoint rather than free metadata.
2426
+ */
2427
+ timestamps?: boolean
2362
2428
  /** The voice to use for generation */
2363
2429
  voice?: string
2364
2430
  /** The output audio format */
@@ -2393,8 +2459,18 @@ export interface TTSResult {
2393
2459
  audio: string
2394
2460
  /** Audio format of the generated audio */
2395
2461
  format: string
2396
- /** Duration of the audio in seconds, if available */
2462
+ /** Duration of the audio file in seconds, if available */
2397
2463
  duration?: number
2464
+ /**
2465
+ * Character- or word-level timings, present when `timestamps: true` was
2466
+ * requested. Use this rather than `duration` to find where *speech* ends.
2467
+ */
2468
+ alignment?: TTSAlignment
2469
+ /**
2470
+ * Per-turn (or per-utterance) spans of the audio, present when
2471
+ * `timestamps: true` was requested and the provider reports segmentation.
2472
+ */
2473
+ segments?: Array<TTSSegment>
2398
2474
  /** Content type of the audio (e.g., 'audio/mp3') */
2399
2475
  contentType?: string
2400
2476
  /** Token usage information (if provided by the adapter) */
@@ -2403,6 +2479,174 @@ export interface TTSResult {
2403
2479
  artifacts?: Array<PersistedArtifactRef>
2404
2480
  }
2405
2481
 
2482
+ // ============================================================================
2483
+ // Voice Catalog Types
2484
+ // ============================================================================
2485
+
2486
+ /**
2487
+ * Where a voice in a provider's catalog came from.
2488
+ *
2489
+ * `'premade'` is the provider's own stock catalog. `'generated'` and
2490
+ * `'cloned'` are voices the account made, which is what `generateVoice()`
2491
+ * produces. `'professional'` covers a provider's curated or paid tiers.
2492
+ */
2493
+ export type VoiceOrigin = 'premade' | 'generated' | 'cloned' | 'professional'
2494
+
2495
+ /** One voice from a provider's catalog. */
2496
+ export interface CatalogVoice {
2497
+ /** Pass this to `generateSpeech()` as `voice` */
2498
+ voiceId: string
2499
+ /** Display name, when the provider stores one */
2500
+ name?: string
2501
+ /** Where the voice came from */
2502
+ origin?: VoiceOrigin
2503
+ /** Provider description of the voice */
2504
+ description?: string
2505
+ /** URL of a sample, when the provider hosts one */
2506
+ previewUrl?: string
2507
+ /** Provider labels, such as accent, age, or use case */
2508
+ labels?: Record<string, string>
2509
+ }
2510
+
2511
+ /** Options for listing a provider's voices. */
2512
+ export interface ListVoicesOptions {
2513
+ /**
2514
+ * Restrict the result to voices of these origins. Adapters filter server
2515
+ * side when the provider supports it, and in memory otherwise.
2516
+ */
2517
+ origins?: Array<VoiceOrigin>
2518
+ /**
2519
+ * Effective abort signal. Adapters forward this to the provider SDK when
2520
+ * supported.
2521
+ */
2522
+ abortSignal?: AbortSignal
2523
+ }
2524
+
2525
+ /** Result of listing a provider's voices. */
2526
+ export interface ListVoicesResult {
2527
+ /** The voices available to this account */
2528
+ voices: Array<CatalogVoice>
2529
+ }
2530
+
2531
+ // ============================================================================
2532
+ // Voice Creation Types
2533
+ // ============================================================================
2534
+
2535
+ /**
2536
+ * Options for creating a voice.
2537
+ *
2538
+ * Providers create voices in one of two ways, and some support both:
2539
+ * - **design** — synthesize a brand new voice from a text {@link prompt}.
2540
+ * - **clone** — derive a voice from {@link referenceAudio} of a real speaker.
2541
+ *
2542
+ * At least one of `prompt` / `referenceAudio` is required; which ones an
2543
+ * adapter accepts depends on the model. An adapter may require both — the
2544
+ * only adapter today, `elevenlabsVoiceDesign`, always needs `prompt` and
2545
+ * takes `referenceAudio` as an additional design reference.
2546
+ */
2547
+ export interface VoiceGenerationOptions<
2548
+ TProviderOptions extends object = object,
2549
+ > {
2550
+ /** The model to use for voice creation */
2551
+ model: string
2552
+ /** Text description of the voice to create, for design-capable models */
2553
+ prompt?: string
2554
+ /**
2555
+ * Reference audio of the speaker to clone - base64 string, base64 data URL,
2556
+ * File, Blob, or ArrayBuffer. For clone-capable models. Remote URLs are not
2557
+ * accepted; read the file and pass the bytes.
2558
+ */
2559
+ referenceAudio?: string | File | Blob | ArrayBuffer
2560
+ /**
2561
+ * Name to store the voice under in the provider's voice library. Providers
2562
+ * differ on what this implies — ElevenLabs only persists a designed voice
2563
+ * when a name is given. Read {@link GeneratedVoice.saved} to find out what
2564
+ * actually happened.
2565
+ */
2566
+ name?: string
2567
+ /** Human-readable description stored alongside the voice */
2568
+ description?: string
2569
+ /** Model-specific options for voice creation */
2570
+ modelOptions?: TProviderOptions
2571
+ /**
2572
+ * Internal logger threaded from the generateVoice() entry point. Adapters
2573
+ * must call logger.request() before the SDK call and logger.errors() in
2574
+ * catch blocks.
2575
+ */
2576
+ logger: InternalLogger
2577
+ /**
2578
+ * Effective abort signal composed by the activity from caller `abortSignal`
2579
+ * and/or `timeout`. Adapters should forward this to the provider SDK when
2580
+ * supported. Request-specific - never store on a global client config.
2581
+ */
2582
+ abortSignal?: AbortSignal
2583
+ }
2584
+
2585
+ /**
2586
+ * A single voice produced by {@link VoiceGenerationOptions}.
2587
+ */
2588
+ export interface GeneratedVoice {
2589
+ /**
2590
+ * The provider's voice identifier. Pass it straight back as the `voice`
2591
+ * option on `generateSpeech()`.
2592
+ */
2593
+ voiceId: string
2594
+ /** Base64-encoded preview audio, when the provider returns one */
2595
+ audio?: string
2596
+ /** Audio format of the preview (e.g. 'mp3') */
2597
+ format?: string
2598
+ /** Content type of the preview (e.g. 'audio/mpeg') */
2599
+ contentType?: string
2600
+ /** Duration of the preview in seconds, if available */
2601
+ duration?: number
2602
+ /** Language of the preview, if reported */
2603
+ language?: string
2604
+ /**
2605
+ * Whether the voice is persisted in the provider's voice library. Unsaved
2606
+ * voices are previews and generally expire.
2607
+ */
2608
+ saved: boolean
2609
+ /**
2610
+ * Whether the voice can be used in `generateSpeech()` yet. Required so a
2611
+ * caller never has to guess: every adapter states it outright.
2612
+ */
2613
+ status: VoiceTrainingStatus
2614
+ }
2615
+
2616
+ /**
2617
+ * Whether a created voice is usable.
2618
+ *
2619
+ * - `'ready'` — usable in `generateSpeech()` now. Every adapter today returns
2620
+ * this, because they all finish the voice inside `generateVoice()`.
2621
+ * - `'training'` — the provider accepted the request but is still building
2622
+ * the voice, so it is not usable yet. Reserved for providers that train
2623
+ * asynchronously; no adapter returns it yet, and reading the state back
2624
+ * will land with the first adapter that needs it.
2625
+ * - `'failed'` — the provider finished without producing a usable voice.
2626
+ */
2627
+ export type VoiceTrainingStatus = 'ready' | 'training' | 'failed'
2628
+
2629
+ /**
2630
+ * Result of voice creation.
2631
+ *
2632
+ * Design models typically return several candidates to choose between; clone
2633
+ * models return exactly one.
2634
+ */
2635
+ export interface VoiceResult {
2636
+ /** Unique identifier for the generation */
2637
+ id: string
2638
+ /** Model used for generation */
2639
+ model: string
2640
+ /** The voices produced, best-first when the provider ranks them */
2641
+ voices: Array<GeneratedVoice>
2642
+ /** The line spoken in the previews, when the provider generated one */
2643
+ previewText?: string
2644
+ /** Token usage information (if provided by the adapter) */
2645
+ usage?: TokenUsage
2646
+ /** Persisted artifact references for generated assets, when available */
2647
+ artifacts?: Array<PersistedArtifactRef>
2648
+ }
2649
+
2406
2650
  // ============================================================================
2407
2651
  // Transcription (Speech-to-Text) Types
2408
2652
  // ============================================================================