@tanstack/ai 0.54.0 → 0.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/README.md +42 -16
  2. package/dist/esm/activities/chat/index.js +17 -1
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/messages.d.ts +21 -1
  5. package/dist/esm/activities/chat/messages.js +50 -1
  6. package/dist/esm/activities/chat/messages.js.map +1 -1
  7. package/dist/esm/activities/chat/stream/processor.d.ts +17 -0
  8. package/dist/esm/activities/chat/stream/processor.js +27 -0
  9. package/dist/esm/activities/chat/stream/processor.js.map +1 -1
  10. package/dist/esm/activities/chat/tools/tool-calls.js +1 -0
  11. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  12. package/dist/esm/activities/evaluate/adapter.d.ts +160 -0
  13. package/dist/esm/activities/evaluate/adapter.js +23 -0
  14. package/dist/esm/activities/evaluate/adapter.js.map +1 -0
  15. package/dist/esm/activities/evaluate/index.d.ts +255 -0
  16. package/dist/esm/activities/evaluate/index.js +317 -0
  17. package/dist/esm/activities/evaluate/index.js.map +1 -0
  18. package/dist/esm/activities/generateSpeech/adapter.d.ts +39 -1
  19. package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
  20. package/dist/esm/activities/generateSpeech/index.d.ts +55 -5
  21. package/dist/esm/activities/generateSpeech/index.js +53 -3
  22. package/dist/esm/activities/generateSpeech/index.js.map +1 -1
  23. package/dist/esm/activities/generateVoice/adapter.d.ts +62 -0
  24. package/dist/esm/activities/generateVoice/adapter.js +23 -0
  25. package/dist/esm/activities/generateVoice/adapter.js.map +1 -0
  26. package/dist/esm/activities/generateVoice/index.d.ts +133 -0
  27. package/dist/esm/activities/generateVoice/index.js +184 -0
  28. package/dist/esm/activities/generateVoice/index.js.map +1 -0
  29. package/dist/esm/activities/index.d.ts +10 -4
  30. package/dist/esm/activities/index.js +14 -10
  31. package/dist/esm/activities/middleware/types.d.ts +1 -1
  32. package/dist/esm/client.d.ts +3 -2
  33. package/dist/esm/client.js +21 -3
  34. package/dist/esm/client.js.map +1 -1
  35. package/dist/esm/index.d.ts +4 -2
  36. package/dist/esm/index.js +5 -2
  37. package/dist/esm/middlewares/otel.js +2 -0
  38. package/dist/esm/middlewares/otel.js.map +1 -1
  39. package/dist/esm/realtime/index.d.ts +1 -1
  40. package/dist/esm/realtime/index.js +1 -1
  41. package/dist/esm/realtime/index.js.map +1 -1
  42. package/dist/esm/types.d.ts +225 -2
  43. package/package.json +2 -2
  44. package/skills/ai-core/adapter-configuration/SKILL.md +1 -1
  45. package/skills/ai-core/media-generation/SKILL.md +154 -18
  46. package/src/activities/chat/index.ts +23 -0
  47. package/src/activities/chat/messages.ts +60 -0
  48. package/src/activities/chat/stream/processor.ts +31 -0
  49. package/src/activities/chat/tools/tool-calls.ts +9 -0
  50. package/src/activities/evaluate/adapter.ts +212 -0
  51. package/src/activities/evaluate/index.ts +614 -0
  52. package/src/activities/generateSpeech/adapter.ts +47 -1
  53. package/src/activities/generateSpeech/index.ts +149 -8
  54. package/src/activities/generateVoice/adapter.ts +89 -0
  55. package/src/activities/generateVoice/index.ts +371 -0
  56. package/src/activities/index.ts +69 -0
  57. package/src/activities/middleware/types.ts +2 -0
  58. package/src/client.ts +35 -8
  59. package/src/index.ts +21 -0
  60. package/src/middlewares/otel.ts +2 -0
  61. package/src/realtime/index.ts +1 -1
  62. package/src/types.ts +246 -2
@@ -26,8 +26,14 @@ import {
26
26
  import type { InternalLogger } from '../../logger/internal-logger'
27
27
  import type { DebugOption } from '../../logger/types'
28
28
  import type { GenerationMiddleware } from '../middleware/types'
29
- import type { TTSAdapter } from './adapter'
30
- import type { StreamChunk, TTSResult } from '../../types'
29
+ import type { TTSAdapter, TTSCapabilities } from './adapter'
30
+ import type {
31
+ ListVoicesOptions,
32
+ ListVoicesResult,
33
+ StreamChunk,
34
+ TTSResult,
35
+ TTSTurn,
36
+ } from '../../types'
31
37
 
32
38
  // ===========================
33
39
  // Activity Kind
@@ -60,16 +66,46 @@ export type TTSProviderOptions<TAdapter> = TAdapter extends {
60
66
  * @template TAdapter - The TTS adapter type
61
67
  * @template TStream - Whether to stream the output
62
68
  */
63
- export interface TTSActivityOptions<
69
+ export type TTSActivityOptions<
70
+ TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
71
+ TStream extends boolean = false,
72
+ > = TTSActivityOptionsBase<TAdapter, TStream> &
73
+ (
74
+ | {
75
+ /** The text to convert to speech */
76
+ text: string
77
+ turns?: undefined
78
+ }
79
+ | {
80
+ text?: undefined
81
+ /**
82
+ * Multi-voice dialogue turns, one per line of the script. Mutually
83
+ * exclusive with `text`.
84
+ *
85
+ * Only adapters that declare `capabilities.maxSpeakers` accept these
86
+ * (ElevenLabs 10 voices, Gemini 2); anything else throws before the
87
+ * request leaves the process.
88
+ */
89
+ turns: Array<TTSTurn>
90
+ }
91
+ )
92
+
93
+ /** Shared half of {@link TTSActivityOptions} — everything except text/turns. */
94
+ interface TTSActivityOptionsBase<
64
95
  TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
65
96
  TStream extends boolean = false,
66
97
  > {
67
98
  /** The TTS adapter to use (must be created with a model) */
68
99
  adapter: TAdapter & { kind: typeof kind }
69
- /** The text to convert to speech */
70
- text: string
71
100
  /** The voice to use for generation */
72
101
  voice?: string
102
+ /**
103
+ * Ask for `alignment` (character/word timings) and `segments` (per-turn
104
+ * spans) on the result. Only adapters that declare
105
+ * `capabilities.timestamps` accept it — on ElevenLabs it is a different
106
+ * endpoint, on BytePlus a different request flag, so it cannot be inferred.
107
+ */
108
+ timestamps?: boolean
73
109
  /** The output audio format */
74
110
  format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'
75
111
  /** The speed of the generated audio (0.25 to 4.0) */
@@ -130,6 +166,57 @@ function createId(prefix: string): string {
130
166
  return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`
131
167
  }
132
168
 
169
+ /**
170
+ * Validate the text/turns/timestamps trio against what the adapter declares,
171
+ * and return the `text` every adapter receives.
172
+ *
173
+ * For a dialogue request that text is the turn scripts joined by newlines:
174
+ * dialogue-aware adapters read `turns` and ignore it, but it keeps `text`
175
+ * non-optional on the adapter contract and gives the devtools event and the
176
+ * artifact inputs something truthful to show.
177
+ */
178
+ function resolveSpeechText(
179
+ adapter: { name: string; capabilities?: TTSCapabilities },
180
+ input: { text?: string; turns?: Array<TTSTurn>; timestamps?: boolean },
181
+ ): string {
182
+ const { text, turns, timestamps } = input
183
+
184
+ if (timestamps && !adapter.capabilities?.timestamps) {
185
+ throw new Error(
186
+ `${adapter.name} cannot return timestamps. Drop \`timestamps: true\` — the result would have no alignment to read.`,
187
+ )
188
+ }
189
+
190
+ if (turns) {
191
+ if (text !== undefined) {
192
+ throw new Error(
193
+ 'generateSpeech() takes either `text` or `turns`, not both.',
194
+ )
195
+ }
196
+ if (turns.length === 0) {
197
+ throw new Error('generateSpeech() `turns` must not be empty.')
198
+ }
199
+ const maxSpeakers = adapter.capabilities?.maxSpeakers
200
+ if (maxSpeakers === undefined) {
201
+ throw new Error(
202
+ `${adapter.name} cannot generate dialogue. Pass \`text\` (and \`voice\`) instead of \`turns\`.`,
203
+ )
204
+ }
205
+ const speakers = new Set(turns.map((turn) => turn.voice)).size
206
+ if (speakers > maxSpeakers) {
207
+ throw new Error(
208
+ `${adapter.name} accepts at most ${maxSpeakers} distinct voice${maxSpeakers === 1 ? '' : 's'} per request; received ${speakers}.`,
209
+ )
210
+ }
211
+ return turns.map((turn) => turn.text).join('\n')
212
+ }
213
+
214
+ if (text === undefined) {
215
+ throw new Error('generateSpeech() requires either `text` or `turns`.')
216
+ }
217
+ return text
218
+ }
219
+
133
220
  // ===========================
134
221
  // Activity Implementation
135
222
  // ===========================
@@ -200,6 +287,7 @@ async function runGenerateSpeech<
200
287
  ...rest
201
288
  } = options
202
289
  const model = adapter.model
290
+ const text = resolveSpeechText(adapter, rest)
203
291
  const requestId = createId('speech')
204
292
  const startTime = Date.now()
205
293
  const logger: InternalLogger = resolveDebugOption(options.debug)
@@ -219,7 +307,7 @@ async function runGenerateSpeech<
219
307
  model,
220
308
  modelOptions: rest.modelOptions,
221
309
  artifactInputs: {
222
- text: rest.text,
310
+ text,
223
311
  voice: rest.voice,
224
312
  format: rest.format,
225
313
  speed: rest.speed,
@@ -235,7 +323,7 @@ async function runGenerateSpeech<
235
323
  requestId,
236
324
  provider: adapter.name,
237
325
  model,
238
- text: rest.text,
326
+ text,
239
327
  voice: rest.voice,
240
328
  format: rest.format,
241
329
  speed: rest.speed,
@@ -252,6 +340,7 @@ async function runGenerateSpeech<
252
340
  const rawResult = await raceWithAbort(
253
341
  adapter.generateSpeech({
254
342
  ...rest,
343
+ text,
255
344
  model,
256
345
  logger,
257
346
  ...(abortControls.signal ? { abortSignal: abortControls.signal } : {}),
@@ -329,6 +418,53 @@ async function runGenerateSpeech<
329
418
  }
330
419
  }
331
420
 
421
+ // ===========================
422
+ // Voice Catalog
423
+ // ===========================
424
+
425
+ /**
426
+ * Options for {@link listVoices}.
427
+ */
428
+ export interface ListVoicesActivityOptions<
429
+ TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
430
+ > extends ListVoicesOptions {
431
+ /** The speech adapter whose catalog to read */
432
+ adapter: TAdapter & { kind: typeof kind }
433
+ }
434
+
435
+ /**
436
+ * List the voices an account can pass to `generateSpeech()`.
437
+ *
438
+ * Only providers with a per-account catalog implement this. A provider whose
439
+ * voices are a fixed list publishes that list as a const in its package, so
440
+ * import it from there rather than calling this.
441
+ *
442
+ * @example Find the voices you created
443
+ * ```ts
444
+ * import { listVoices } from '@tanstack/ai'
445
+ * import { elevenlabsSpeech } from '@tanstack/ai-elevenlabs'
446
+ *
447
+ * const { voices } = await listVoices({
448
+ * adapter: elevenlabsSpeech('eleven_v3'),
449
+ * origins: ['generated', 'cloned'],
450
+ * })
451
+ * ```
452
+ */
453
+ export async function listVoices<
454
+ TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
455
+ >(options: ListVoicesActivityOptions<TAdapter>): Promise<ListVoicesResult> {
456
+ const { adapter, ...rest } = options
457
+
458
+ const list = adapter.listVoices
459
+ if (!list) {
460
+ throw new Error(
461
+ `The ${adapter.name} speech adapter has no per-account voice catalog to list. Its voices are a fixed set — import the voice list or union its package exports instead (for example \`GeminiTTSVoices\` from @tanstack/ai-gemini, or the \`OpenAITTSVoice\` union from @tanstack/ai-openai).`,
462
+ )
463
+ }
464
+
465
+ return await list.call(adapter, rest)
466
+ }
467
+
332
468
  // ===========================
333
469
  // Options Factory
334
470
  // ===========================
@@ -346,5 +482,10 @@ export function createSpeechOptions<
346
482
  }
347
483
 
348
484
  // Re-export adapter types
349
- export type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'
485
+ export type {
486
+ TTSAdapter,
487
+ TTSAdapterConfig,
488
+ TTSCapabilities,
489
+ AnyTTSAdapter,
490
+ } from './adapter'
350
491
  export { BaseTTSAdapter } from './adapter'
@@ -0,0 +1,89 @@
1
+ import type { VoiceGenerationOptions, VoiceResult } from '../../types'
2
+
3
+ /**
4
+ * Configuration for voice adapter instances
5
+ */
6
+ export interface VoiceAdapterConfig {
7
+ apiKey?: string
8
+ baseUrl?: string
9
+ timeout?: number
10
+ maxRetries?: number
11
+ headers?: Record<string, string>
12
+ }
13
+
14
+ /**
15
+ * Voice adapter interface with pre-resolved generics.
16
+ *
17
+ * An adapter is created by a provider function: `provider('model')` → `adapter`
18
+ * All type resolution happens at the provider call site, not in this interface.
19
+ *
20
+ * Generic parameters:
21
+ * - TModel: The specific model name (e.g., 'eleven_ttv_v3')
22
+ * - TProviderOptions: Provider-specific options (already resolved)
23
+ */
24
+ export interface VoiceAdapter<
25
+ TModel extends string = string,
26
+ TProviderOptions extends object = Record<string, unknown>,
27
+ > {
28
+ /** Discriminator for adapter kind - used to determine API shape */
29
+ readonly kind: 'voice'
30
+ /** Adapter name identifier */
31
+ readonly name: string
32
+ /** The model this adapter is configured for */
33
+ readonly model: TModel
34
+
35
+ /**
36
+ * @internal Type-only properties for inference. Not assigned at runtime.
37
+ */
38
+ '~types': {
39
+ providerOptions: TProviderOptions
40
+ }
41
+
42
+ /**
43
+ * Create a voice from a text description and/or reference audio
44
+ */
45
+ generateVoice: (
46
+ options: VoiceGenerationOptions<TProviderOptions>,
47
+ ) => Promise<VoiceResult>
48
+ }
49
+
50
+ /**
51
+ * A VoiceAdapter with any/unknown type parameters.
52
+ * Useful as a constraint in generic functions and interfaces.
53
+ */
54
+ export type AnyVoiceAdapter = VoiceAdapter<any, any>
55
+
56
+ /**
57
+ * Abstract base class for voice creation adapters.
58
+ * Extend this class to implement a voice adapter for a specific provider.
59
+ *
60
+ * Generic parameters match VoiceAdapter - all pre-resolved by the provider function.
61
+ */
62
+ export abstract class BaseVoiceAdapter<
63
+ TModel extends string = string,
64
+ TProviderOptions extends object = Record<string, unknown>,
65
+ > implements VoiceAdapter<TModel, TProviderOptions> {
66
+ readonly kind = 'voice' as const
67
+ abstract readonly name: string
68
+ readonly model: TModel
69
+
70
+ // Type-only property - never assigned at runtime
71
+ declare '~types': {
72
+ providerOptions: TProviderOptions
73
+ }
74
+
75
+ protected config: VoiceAdapterConfig
76
+
77
+ constructor(model: TModel, config: VoiceAdapterConfig = {}) {
78
+ this.config = config
79
+ this.model = model
80
+ }
81
+
82
+ abstract generateVoice(
83
+ options: VoiceGenerationOptions<TProviderOptions>,
84
+ ): Promise<VoiceResult>
85
+
86
+ protected generateId(): string {
87
+ return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`
88
+ }
89
+ }
@@ -0,0 +1,371 @@
1
+ /**
2
+ * Voice Activity
3
+ *
4
+ * Creates a reusable voice — either designed from a text description or cloned
5
+ * from reference audio — and returns voice ids that `generateSpeech()` accepts.
6
+ * This is a self-contained module with implementation, types, and JSDoc.
7
+ */
8
+
9
+ import { aiEventClient } from '@tanstack/ai-event-client'
10
+ import { streamGenerationResult } from '../stream-generation-result.js'
11
+ import { resolveDebugOption } from '../../logger/resolve'
12
+ import {
13
+ applyGenerationResultTransforms,
14
+ createGenerationContext,
15
+ runGenerationAbort,
16
+ runGenerationError,
17
+ runGenerationFinish,
18
+ runGenerationStart,
19
+ runGenerationUsage,
20
+ } from '../middleware/run'
21
+ import {
22
+ abortReasonMessage,
23
+ createActivityAbortControls,
24
+ isActivityAbortError,
25
+ raceWithAbort,
26
+ } from '../../utilities/activity-abort'
27
+ import type { InternalLogger } from '../../logger/internal-logger'
28
+ import type { DebugOption } from '../../logger/types'
29
+ import type { GenerationMiddleware } from '../middleware/types'
30
+ import type { VoiceAdapter } from './adapter'
31
+ import type { StreamChunk, VoiceResult } from '../../types'
32
+
33
+ // ===========================
34
+ // Activity Kind
35
+ // ===========================
36
+
37
+ /** The adapter kind this activity handles */
38
+ export const kind = 'voice' as const
39
+
40
+ // ===========================
41
+ // Type Extraction Helpers
42
+ // ===========================
43
+
44
+ /**
45
+ * Extract provider options from a VoiceAdapter via ~types.
46
+ */
47
+ export type VoiceProviderOptions<TAdapter> = TAdapter extends {
48
+ '~types': { providerOptions: infer P extends object }
49
+ }
50
+ ? P
51
+ : object
52
+
53
+ // ===========================
54
+ // Activity Options Type
55
+ // ===========================
56
+
57
+ /**
58
+ * Options for the voice activity.
59
+ * The model is extracted from the adapter's model property.
60
+ *
61
+ * @template TAdapter - The voice adapter type
62
+ * @template TStream - Whether to stream the output
63
+ */
64
+ export interface VoiceActivityOptions<
65
+ TAdapter extends VoiceAdapter<string, VoiceProviderOptions<TAdapter>>,
66
+ TStream extends boolean = false,
67
+ > {
68
+ /** The voice adapter to use (must be created with a model) */
69
+ adapter: TAdapter & { kind: typeof kind }
70
+ /**
71
+ * Text description of the voice to create, for design-capable models
72
+ * (e.g. `'A warm, gravelly narrator in his sixties'`).
73
+ */
74
+ prompt?: string
75
+ /**
76
+ * Reference audio of the speaker to clone, for clone-capable models.
77
+ * Accepts a base64 string, base64 data URL, File, Blob, or ArrayBuffer.
78
+ * Remote URLs are not accepted; read the file and pass the bytes.
79
+ */
80
+ referenceAudio?: string | File | Blob | ArrayBuffer
81
+ /**
82
+ * Name to store the voice under in the provider's voice library. Check
83
+ * `saved` on each returned voice to see whether it was actually persisted.
84
+ */
85
+ name?: string
86
+ /** Human-readable description stored alongside the voice */
87
+ description?: string
88
+ /** Provider-specific options for voice creation */
89
+ modelOptions?: VoiceProviderOptions<TAdapter>
90
+ /**
91
+ * Whether to stream the generation result.
92
+ * When true, returns an AsyncIterable<StreamChunk> for streaming transport.
93
+ * When false or not provided, returns a Promise<VoiceResult>.
94
+ *
95
+ * @default false
96
+ */
97
+ stream?: TStream
98
+ /**
99
+ * Enable debug logging. Pass `true` to enable all categories, `false` to
100
+ * silence everything including errors, or a `DebugConfig` object for granular
101
+ * control and/or a custom `Logger`.
102
+ */
103
+ debug?: DebugOption
104
+ /**
105
+ * Observe-only middleware notified on start, usage, success, and error. Pass
106
+ * `otelMiddleware()` to emit OpenTelemetry spans, or implement the
107
+ * `GenerationMiddleware` contract for a custom backend.
108
+ */
109
+ middleware?: Array<GenerationMiddleware>
110
+ /** Stable conversation/thread id for correlating this run when persisted. */
111
+ threadId?: string
112
+ /** Stable run id for correlating this run when persisted. */
113
+ runId?: string
114
+ /**
115
+ * Maximum duration of this activity invocation in milliseconds.
116
+ * No SDK-wide default — choose a value suitable for the provider and job.
117
+ * Composed with {@link abortSignal}; the first abort wins.
118
+ */
119
+ timeout?: number
120
+ /**
121
+ * Caller cancellation signal (request disconnects, job/runtime cancellation).
122
+ * Composed with {@link timeout} into an effective signal forwarded to the
123
+ * adapter. Request-specific — not stored on global provider client config.
124
+ */
125
+ abortSignal?: AbortSignal
126
+ }
127
+
128
+ // ===========================
129
+ // Activity Result Type
130
+ // ===========================
131
+
132
+ /**
133
+ * Result type for the voice activity.
134
+ * - If stream is true: AsyncIterable<StreamChunk>
135
+ * - Otherwise: Promise<VoiceResult>
136
+ */
137
+ export type VoiceActivityResult<TStream extends boolean = false> =
138
+ TStream extends true ? AsyncIterable<StreamChunk> : Promise<VoiceResult>
139
+
140
+ function createId(prefix: string): string {
141
+ return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`
142
+ }
143
+
144
+ // ===========================
145
+ // Activity Implementation
146
+ // ===========================
147
+
148
+ /**
149
+ * Voice activity - creates a reusable voice.
150
+ *
151
+ * Providers create voices in one of two ways, and some support both: design a
152
+ * new voice from a text description, or clone one from reference audio. Either
153
+ * way the result carries voice ids you pass back to `generateSpeech()`.
154
+ *
155
+ * @example Design a voice from a description
156
+ * ```ts
157
+ * import { generateVoice, generateSpeech } from '@tanstack/ai'
158
+ * import { elevenlabsVoiceDesign, elevenlabsSpeech } from '@tanstack/ai-elevenlabs'
159
+ *
160
+ * const designed = await generateVoice({
161
+ * adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
162
+ * prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',
163
+ * })
164
+ *
165
+ * const [preview] = designed.voices
166
+ * if (!preview) throw new Error('No voice candidates returned')
167
+ *
168
+ * const speech = await generateSpeech({
169
+ * adapter: elevenlabsSpeech('eleven_v3'),
170
+ * text: 'Once upon a time...',
171
+ * voice: preview.voiceId,
172
+ * })
173
+ * ```
174
+ *
175
+ * @example Save the voice to the provider's library
176
+ * ```ts
177
+ * const saved = await generateVoice({
178
+ * adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
179
+ * prompt: 'A bright, upbeat product demo host',
180
+ * name: 'Demo Host',
181
+ * description: 'Bright, upbeat, mid-30s',
182
+ * })
183
+ * ```
184
+ */
185
+ export function generateVoice<
186
+ TAdapter extends VoiceAdapter<string, VoiceProviderOptions<TAdapter>>,
187
+ TStream extends boolean = false,
188
+ >(
189
+ options: VoiceActivityOptions<TAdapter, TStream>,
190
+ ): VoiceActivityResult<TStream> {
191
+ if (options.stream) {
192
+ return streamGenerationResult(
193
+ // Only `runId` is taken from the resolved wire identity — see the
194
+ // matching note in `generateSpeech`.
195
+ (resolved) => runGenerateVoice({ ...options, runId: resolved.runId }),
196
+ options,
197
+ ) as VoiceActivityResult<TStream>
198
+ }
199
+ return runGenerateVoice(options) as VoiceActivityResult<TStream>
200
+ }
201
+
202
+ /**
203
+ * Run the core voice generation logic (non-streaming).
204
+ */
205
+ async function runGenerateVoice<
206
+ TAdapter extends VoiceAdapter<string, VoiceProviderOptions<TAdapter>>,
207
+ >(options: VoiceActivityOptions<TAdapter, boolean>): Promise<VoiceResult> {
208
+ const {
209
+ adapter,
210
+ stream: _stream,
211
+ debug: _debug,
212
+ middleware,
213
+ threadId,
214
+ runId,
215
+ timeout,
216
+ abortSignal: callerAbortSignal,
217
+ ...rest
218
+ } = options
219
+
220
+ if (rest.prompt == null && rest.referenceAudio == null) {
221
+ throw new Error(
222
+ 'generateVoice() requires `prompt` (design a new voice) or `referenceAudio` (clone an existing one).',
223
+ )
224
+ }
225
+
226
+ const model = adapter.model
227
+ const requestId = createId('voice')
228
+ const startTime = Date.now()
229
+ const logger: InternalLogger = resolveDebugOption(options.debug)
230
+ const abortControls = createActivityAbortControls({
231
+ timeout,
232
+ abortSignal: callerAbortSignal,
233
+ })
234
+
235
+ const mwCtx = createGenerationContext({
236
+ requestId,
237
+ activity: 'voice',
238
+ provider: adapter.name,
239
+ model,
240
+ modelOptions: rest.modelOptions,
241
+ artifactInputs: {
242
+ prompt: rest.prompt,
243
+ name: rest.name,
244
+ description: rest.description,
245
+ },
246
+ threadId,
247
+ runId,
248
+ createId,
249
+ })
250
+
251
+ await runGenerationStart(middleware, mwCtx)
252
+
253
+ aiEventClient.emit('voice:request:started', {
254
+ requestId,
255
+ provider: adapter.name,
256
+ model,
257
+ prompt: rest.prompt,
258
+ name: rest.name,
259
+ description: rest.description,
260
+ hasReferenceAudio: rest.referenceAudio != null,
261
+ modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
262
+ timestamp: startTime,
263
+ })
264
+
265
+ logger.request(`activity=generateVoice provider=${adapter.name}`, {
266
+ provider: adapter.name,
267
+ model,
268
+ })
269
+
270
+ try {
271
+ const rawResult = await raceWithAbort(
272
+ adapter.generateVoice({
273
+ ...rest,
274
+ model,
275
+ logger,
276
+ ...(abortControls.signal ? { abortSignal: abortControls.signal } : {}),
277
+ }),
278
+ abortControls.signal,
279
+ )
280
+ abortControls.clear()
281
+ const result = await applyGenerationResultTransforms(mwCtx, rawResult)
282
+ const duration = Date.now() - startTime
283
+
284
+ aiEventClient.emit('voice:request:completed', {
285
+ requestId,
286
+ provider: adapter.name,
287
+ model,
288
+ voiceIds: result.voices.map((voice) => voice.voiceId),
289
+ voiceCount: result.voices.length,
290
+ previewText: result.previewText,
291
+ duration,
292
+ modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
293
+ timestamp: Date.now(),
294
+ })
295
+
296
+ if (result.usage) {
297
+ aiEventClient.emit('voice:usage', {
298
+ requestId,
299
+ model,
300
+ usage: result.usage,
301
+ modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
302
+ timestamp: Date.now(),
303
+ })
304
+ }
305
+
306
+ logger.output(`activity=generateVoice voices=${result.voices.length}`, {
307
+ voices: result.voices.length,
308
+ })
309
+
310
+ if (result.usage) await runGenerationUsage(middleware, mwCtx, result.usage)
311
+ await runGenerationFinish(middleware, mwCtx, {
312
+ duration,
313
+ usage: result.usage,
314
+ })
315
+
316
+ return result
317
+ } catch (error) {
318
+ abortControls.clear()
319
+ const duration = Date.now() - startTime
320
+ const err = error as Error
321
+ aiEventClient.emit('voice:request:error', {
322
+ requestId,
323
+ provider: adapter.name,
324
+ model,
325
+ error: { message: err.message, name: err.name },
326
+ duration,
327
+ modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
328
+ timestamp: Date.now(),
329
+ })
330
+ if (isActivityAbortError(error, abortControls.signal)) {
331
+ await runGenerationAbort(middleware, mwCtx, {
332
+ reason: abortReasonMessage(error, abortControls.signal),
333
+ duration,
334
+ })
335
+ } else {
336
+ await runGenerationError(middleware, mwCtx, {
337
+ error,
338
+ duration,
339
+ })
340
+ }
341
+ logger.errors('generateVoice activity failed', {
342
+ error,
343
+ source: 'generateVoice',
344
+ })
345
+ throw error
346
+ }
347
+ }
348
+
349
+ // ===========================
350
+ // Options Factory
351
+ // ===========================
352
+
353
+ /**
354
+ * Create typed options for the generateVoice() function without executing.
355
+ */
356
+ export function createVoiceOptions<
357
+ TAdapter extends VoiceAdapter<string, VoiceProviderOptions<TAdapter>>,
358
+ TStream extends boolean = false,
359
+ >(
360
+ options: VoiceActivityOptions<TAdapter, TStream>,
361
+ ): VoiceActivityOptions<TAdapter, TStream> {
362
+ return options
363
+ }
364
+
365
+ // Re-export adapter types
366
+ export type {
367
+ VoiceAdapter,
368
+ VoiceAdapterConfig,
369
+ AnyVoiceAdapter,
370
+ } from './adapter'
371
+ export { BaseVoiceAdapter } from './adapter'