@tanstack/ai 0.55.0 → 0.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +42 -16
- package/dist/esm/activities/chat/tools/tool-calls.js +1 -0
- package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
- package/dist/esm/activities/evaluate/adapter.d.ts +160 -0
- package/dist/esm/activities/evaluate/adapter.js +23 -0
- package/dist/esm/activities/evaluate/adapter.js.map +1 -0
- package/dist/esm/activities/evaluate/index.d.ts +255 -0
- package/dist/esm/activities/evaluate/index.js +317 -0
- package/dist/esm/activities/evaluate/index.js.map +1 -0
- package/dist/esm/activities/generateSpeech/adapter.d.ts +39 -1
- package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +55 -5
- package/dist/esm/activities/generateSpeech/index.js +53 -3
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/activities/generateVoice/adapter.d.ts +62 -0
- package/dist/esm/activities/generateVoice/adapter.js +23 -0
- package/dist/esm/activities/generateVoice/adapter.js.map +1 -0
- package/dist/esm/activities/generateVoice/index.d.ts +133 -0
- package/dist/esm/activities/generateVoice/index.js +184 -0
- package/dist/esm/activities/generateVoice/index.js.map +1 -0
- package/dist/esm/activities/index.d.ts +10 -4
- package/dist/esm/activities/index.js +14 -10
- package/dist/esm/activities/middleware/types.d.ts +1 -1
- package/dist/esm/client.d.ts +3 -2
- package/dist/esm/client.js +21 -3
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +4 -2
- package/dist/esm/index.js +5 -2
- package/dist/esm/middlewares/otel.js +2 -0
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/realtime/index.d.ts +1 -1
- package/dist/esm/realtime/index.js +1 -1
- package/dist/esm/realtime/index.js.map +1 -1
- package/dist/esm/types.d.ts +225 -2
- package/package.json +3 -3
- package/skills/ai-core/media-generation/SKILL.md +132 -6
- package/src/activities/chat/tools/tool-calls.ts +9 -0
- package/src/activities/evaluate/adapter.ts +212 -0
- package/src/activities/evaluate/index.ts +614 -0
- package/src/activities/generateSpeech/adapter.ts +47 -1
- package/src/activities/generateSpeech/index.ts +149 -8
- package/src/activities/generateVoice/adapter.ts +89 -0
- package/src/activities/generateVoice/index.ts +371 -0
- package/src/activities/index.ts +69 -0
- package/src/activities/middleware/types.ts +2 -0
- package/src/client.ts +35 -8
- package/src/index.ts +21 -0
- package/src/middlewares/otel.ts +2 -0
- package/src/realtime/index.ts +1 -1
- package/src/types.ts +246 -2
|
@@ -26,8 +26,14 @@ import {
|
|
|
26
26
|
import type { InternalLogger } from '../../logger/internal-logger'
|
|
27
27
|
import type { DebugOption } from '../../logger/types'
|
|
28
28
|
import type { GenerationMiddleware } from '../middleware/types'
|
|
29
|
-
import type { TTSAdapter } from './adapter'
|
|
30
|
-
import type {
|
|
29
|
+
import type { TTSAdapter, TTSCapabilities } from './adapter'
|
|
30
|
+
import type {
|
|
31
|
+
ListVoicesOptions,
|
|
32
|
+
ListVoicesResult,
|
|
33
|
+
StreamChunk,
|
|
34
|
+
TTSResult,
|
|
35
|
+
TTSTurn,
|
|
36
|
+
} from '../../types'
|
|
31
37
|
|
|
32
38
|
// ===========================
|
|
33
39
|
// Activity Kind
|
|
@@ -60,16 +66,46 @@ export type TTSProviderOptions<TAdapter> = TAdapter extends {
|
|
|
60
66
|
* @template TAdapter - The TTS adapter type
|
|
61
67
|
* @template TStream - Whether to stream the output
|
|
62
68
|
*/
|
|
63
|
-
export
|
|
69
|
+
export type TTSActivityOptions<
|
|
70
|
+
TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
|
|
71
|
+
TStream extends boolean = false,
|
|
72
|
+
> = TTSActivityOptionsBase<TAdapter, TStream> &
|
|
73
|
+
(
|
|
74
|
+
| {
|
|
75
|
+
/** The text to convert to speech */
|
|
76
|
+
text: string
|
|
77
|
+
turns?: undefined
|
|
78
|
+
}
|
|
79
|
+
| {
|
|
80
|
+
text?: undefined
|
|
81
|
+
/**
|
|
82
|
+
* Multi-voice dialogue turns, one per line of the script. Mutually
|
|
83
|
+
* exclusive with `text`.
|
|
84
|
+
*
|
|
85
|
+
* Only adapters that declare `capabilities.maxSpeakers` accept these
|
|
86
|
+
* (ElevenLabs 10 voices, Gemini 2); anything else throws before the
|
|
87
|
+
* request leaves the process.
|
|
88
|
+
*/
|
|
89
|
+
turns: Array<TTSTurn>
|
|
90
|
+
}
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
/** Shared half of {@link TTSActivityOptions} — everything except text/turns. */
|
|
94
|
+
interface TTSActivityOptionsBase<
|
|
64
95
|
TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
|
|
65
96
|
TStream extends boolean = false,
|
|
66
97
|
> {
|
|
67
98
|
/** The TTS adapter to use (must be created with a model) */
|
|
68
99
|
adapter: TAdapter & { kind: typeof kind }
|
|
69
|
-
/** The text to convert to speech */
|
|
70
|
-
text: string
|
|
71
100
|
/** The voice to use for generation */
|
|
72
101
|
voice?: string
|
|
102
|
+
/**
|
|
103
|
+
* Ask for `alignment` (character/word timings) and `segments` (per-turn
|
|
104
|
+
* spans) on the result. Only adapters that declare
|
|
105
|
+
* `capabilities.timestamps` accept it — on ElevenLabs it is a different
|
|
106
|
+
* endpoint, on BytePlus a different request flag, so it cannot be inferred.
|
|
107
|
+
*/
|
|
108
|
+
timestamps?: boolean
|
|
73
109
|
/** The output audio format */
|
|
74
110
|
format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'
|
|
75
111
|
/** The speed of the generated audio (0.25 to 4.0) */
|
|
@@ -130,6 +166,57 @@ function createId(prefix: string): string {
|
|
|
130
166
|
return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`
|
|
131
167
|
}
|
|
132
168
|
|
|
169
|
+
/**
|
|
170
|
+
* Validate the text/turns/timestamps trio against what the adapter declares,
|
|
171
|
+
* and return the `text` every adapter receives.
|
|
172
|
+
*
|
|
173
|
+
* For a dialogue request that text is the turn scripts joined by newlines:
|
|
174
|
+
* dialogue-aware adapters read `turns` and ignore it, but it keeps `text`
|
|
175
|
+
* non-optional on the adapter contract and gives the devtools event and the
|
|
176
|
+
* artifact inputs something truthful to show.
|
|
177
|
+
*/
|
|
178
|
+
function resolveSpeechText(
|
|
179
|
+
adapter: { name: string; capabilities?: TTSCapabilities },
|
|
180
|
+
input: { text?: string; turns?: Array<TTSTurn>; timestamps?: boolean },
|
|
181
|
+
): string {
|
|
182
|
+
const { text, turns, timestamps } = input
|
|
183
|
+
|
|
184
|
+
if (timestamps && !adapter.capabilities?.timestamps) {
|
|
185
|
+
throw new Error(
|
|
186
|
+
`${adapter.name} cannot return timestamps. Drop \`timestamps: true\` — the result would have no alignment to read.`,
|
|
187
|
+
)
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
if (turns) {
|
|
191
|
+
if (text !== undefined) {
|
|
192
|
+
throw new Error(
|
|
193
|
+
'generateSpeech() takes either `text` or `turns`, not both.',
|
|
194
|
+
)
|
|
195
|
+
}
|
|
196
|
+
if (turns.length === 0) {
|
|
197
|
+
throw new Error('generateSpeech() `turns` must not be empty.')
|
|
198
|
+
}
|
|
199
|
+
const maxSpeakers = adapter.capabilities?.maxSpeakers
|
|
200
|
+
if (maxSpeakers === undefined) {
|
|
201
|
+
throw new Error(
|
|
202
|
+
`${adapter.name} cannot generate dialogue. Pass \`text\` (and \`voice\`) instead of \`turns\`.`,
|
|
203
|
+
)
|
|
204
|
+
}
|
|
205
|
+
const speakers = new Set(turns.map((turn) => turn.voice)).size
|
|
206
|
+
if (speakers > maxSpeakers) {
|
|
207
|
+
throw new Error(
|
|
208
|
+
`${adapter.name} accepts at most ${maxSpeakers} distinct voice${maxSpeakers === 1 ? '' : 's'} per request; received ${speakers}.`,
|
|
209
|
+
)
|
|
210
|
+
}
|
|
211
|
+
return turns.map((turn) => turn.text).join('\n')
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
if (text === undefined) {
|
|
215
|
+
throw new Error('generateSpeech() requires either `text` or `turns`.')
|
|
216
|
+
}
|
|
217
|
+
return text
|
|
218
|
+
}
|
|
219
|
+
|
|
133
220
|
// ===========================
|
|
134
221
|
// Activity Implementation
|
|
135
222
|
// ===========================
|
|
@@ -200,6 +287,7 @@ async function runGenerateSpeech<
|
|
|
200
287
|
...rest
|
|
201
288
|
} = options
|
|
202
289
|
const model = adapter.model
|
|
290
|
+
const text = resolveSpeechText(adapter, rest)
|
|
203
291
|
const requestId = createId('speech')
|
|
204
292
|
const startTime = Date.now()
|
|
205
293
|
const logger: InternalLogger = resolveDebugOption(options.debug)
|
|
@@ -219,7 +307,7 @@ async function runGenerateSpeech<
|
|
|
219
307
|
model,
|
|
220
308
|
modelOptions: rest.modelOptions,
|
|
221
309
|
artifactInputs: {
|
|
222
|
-
text
|
|
310
|
+
text,
|
|
223
311
|
voice: rest.voice,
|
|
224
312
|
format: rest.format,
|
|
225
313
|
speed: rest.speed,
|
|
@@ -235,7 +323,7 @@ async function runGenerateSpeech<
|
|
|
235
323
|
requestId,
|
|
236
324
|
provider: adapter.name,
|
|
237
325
|
model,
|
|
238
|
-
text
|
|
326
|
+
text,
|
|
239
327
|
voice: rest.voice,
|
|
240
328
|
format: rest.format,
|
|
241
329
|
speed: rest.speed,
|
|
@@ -252,6 +340,7 @@ async function runGenerateSpeech<
|
|
|
252
340
|
const rawResult = await raceWithAbort(
|
|
253
341
|
adapter.generateSpeech({
|
|
254
342
|
...rest,
|
|
343
|
+
text,
|
|
255
344
|
model,
|
|
256
345
|
logger,
|
|
257
346
|
...(abortControls.signal ? { abortSignal: abortControls.signal } : {}),
|
|
@@ -329,6 +418,53 @@ async function runGenerateSpeech<
|
|
|
329
418
|
}
|
|
330
419
|
}
|
|
331
420
|
|
|
421
|
+
// ===========================
|
|
422
|
+
// Voice Catalog
|
|
423
|
+
// ===========================
|
|
424
|
+
|
|
425
|
+
/**
|
|
426
|
+
* Options for {@link listVoices}.
|
|
427
|
+
*/
|
|
428
|
+
export interface ListVoicesActivityOptions<
|
|
429
|
+
TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
|
|
430
|
+
> extends ListVoicesOptions {
|
|
431
|
+
/** The speech adapter whose catalog to read */
|
|
432
|
+
adapter: TAdapter & { kind: typeof kind }
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
/**
|
|
436
|
+
* List the voices an account can pass to `generateSpeech()`.
|
|
437
|
+
*
|
|
438
|
+
* Only providers with a per-account catalog implement this. A provider whose
|
|
439
|
+
* voices are a fixed list publishes that list as a const in its package, so
|
|
440
|
+
* import it from there rather than calling this.
|
|
441
|
+
*
|
|
442
|
+
* @example Find the voices you created
|
|
443
|
+
* ```ts
|
|
444
|
+
* import { listVoices } from '@tanstack/ai'
|
|
445
|
+
* import { elevenlabsSpeech } from '@tanstack/ai-elevenlabs'
|
|
446
|
+
*
|
|
447
|
+
* const { voices } = await listVoices({
|
|
448
|
+
* adapter: elevenlabsSpeech('eleven_v3'),
|
|
449
|
+
* origins: ['generated', 'cloned'],
|
|
450
|
+
* })
|
|
451
|
+
* ```
|
|
452
|
+
*/
|
|
453
|
+
export async function listVoices<
|
|
454
|
+
TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
|
|
455
|
+
>(options: ListVoicesActivityOptions<TAdapter>): Promise<ListVoicesResult> {
|
|
456
|
+
const { adapter, ...rest } = options
|
|
457
|
+
|
|
458
|
+
const list = adapter.listVoices
|
|
459
|
+
if (!list) {
|
|
460
|
+
throw new Error(
|
|
461
|
+
`The ${adapter.name} speech adapter has no per-account voice catalog to list. Its voices are a fixed set — import the voice list or union its package exports instead (for example \`GeminiTTSVoices\` from @tanstack/ai-gemini, or the \`OpenAITTSVoice\` union from @tanstack/ai-openai).`,
|
|
462
|
+
)
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
return await list.call(adapter, rest)
|
|
466
|
+
}
|
|
467
|
+
|
|
332
468
|
// ===========================
|
|
333
469
|
// Options Factory
|
|
334
470
|
// ===========================
|
|
@@ -346,5 +482,10 @@ export function createSpeechOptions<
|
|
|
346
482
|
}
|
|
347
483
|
|
|
348
484
|
// Re-export adapter types
|
|
349
|
-
export type {
|
|
485
|
+
export type {
|
|
486
|
+
TTSAdapter,
|
|
487
|
+
TTSAdapterConfig,
|
|
488
|
+
TTSCapabilities,
|
|
489
|
+
AnyTTSAdapter,
|
|
490
|
+
} from './adapter'
|
|
350
491
|
export { BaseTTSAdapter } from './adapter'
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
import type { VoiceGenerationOptions, VoiceResult } from '../../types'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Configuration for voice adapter instances
|
|
5
|
+
*/
|
|
6
|
+
export interface VoiceAdapterConfig {
|
|
7
|
+
apiKey?: string
|
|
8
|
+
baseUrl?: string
|
|
9
|
+
timeout?: number
|
|
10
|
+
maxRetries?: number
|
|
11
|
+
headers?: Record<string, string>
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Voice adapter interface with pre-resolved generics.
|
|
16
|
+
*
|
|
17
|
+
* An adapter is created by a provider function: `provider('model')` → `adapter`
|
|
18
|
+
* All type resolution happens at the provider call site, not in this interface.
|
|
19
|
+
*
|
|
20
|
+
* Generic parameters:
|
|
21
|
+
* - TModel: The specific model name (e.g., 'eleven_ttv_v3')
|
|
22
|
+
* - TProviderOptions: Provider-specific options (already resolved)
|
|
23
|
+
*/
|
|
24
|
+
export interface VoiceAdapter<
|
|
25
|
+
TModel extends string = string,
|
|
26
|
+
TProviderOptions extends object = Record<string, unknown>,
|
|
27
|
+
> {
|
|
28
|
+
/** Discriminator for adapter kind - used to determine API shape */
|
|
29
|
+
readonly kind: 'voice'
|
|
30
|
+
/** Adapter name identifier */
|
|
31
|
+
readonly name: string
|
|
32
|
+
/** The model this adapter is configured for */
|
|
33
|
+
readonly model: TModel
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* @internal Type-only properties for inference. Not assigned at runtime.
|
|
37
|
+
*/
|
|
38
|
+
'~types': {
|
|
39
|
+
providerOptions: TProviderOptions
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Create a voice from a text description and/or reference audio
|
|
44
|
+
*/
|
|
45
|
+
generateVoice: (
|
|
46
|
+
options: VoiceGenerationOptions<TProviderOptions>,
|
|
47
|
+
) => Promise<VoiceResult>
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* A VoiceAdapter with any/unknown type parameters.
|
|
52
|
+
* Useful as a constraint in generic functions and interfaces.
|
|
53
|
+
*/
|
|
54
|
+
export type AnyVoiceAdapter = VoiceAdapter<any, any>
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Abstract base class for voice creation adapters.
|
|
58
|
+
* Extend this class to implement a voice adapter for a specific provider.
|
|
59
|
+
*
|
|
60
|
+
* Generic parameters match VoiceAdapter - all pre-resolved by the provider function.
|
|
61
|
+
*/
|
|
62
|
+
export abstract class BaseVoiceAdapter<
|
|
63
|
+
TModel extends string = string,
|
|
64
|
+
TProviderOptions extends object = Record<string, unknown>,
|
|
65
|
+
> implements VoiceAdapter<TModel, TProviderOptions> {
|
|
66
|
+
readonly kind = 'voice' as const
|
|
67
|
+
abstract readonly name: string
|
|
68
|
+
readonly model: TModel
|
|
69
|
+
|
|
70
|
+
// Type-only property - never assigned at runtime
|
|
71
|
+
declare '~types': {
|
|
72
|
+
providerOptions: TProviderOptions
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
protected config: VoiceAdapterConfig
|
|
76
|
+
|
|
77
|
+
constructor(model: TModel, config: VoiceAdapterConfig = {}) {
|
|
78
|
+
this.config = config
|
|
79
|
+
this.model = model
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
abstract generateVoice(
|
|
83
|
+
options: VoiceGenerationOptions<TProviderOptions>,
|
|
84
|
+
): Promise<VoiceResult>
|
|
85
|
+
|
|
86
|
+
protected generateId(): string {
|
|
87
|
+
return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`
|
|
88
|
+
}
|
|
89
|
+
}
|
|
@@ -0,0 +1,371 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Voice Activity
|
|
3
|
+
*
|
|
4
|
+
* Creates a reusable voice — either designed from a text description or cloned
|
|
5
|
+
* from reference audio — and returns voice ids that `generateSpeech()` accepts.
|
|
6
|
+
* This is a self-contained module with implementation, types, and JSDoc.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { aiEventClient } from '@tanstack/ai-event-client'
|
|
10
|
+
import { streamGenerationResult } from '../stream-generation-result.js'
|
|
11
|
+
import { resolveDebugOption } from '../../logger/resolve'
|
|
12
|
+
import {
|
|
13
|
+
applyGenerationResultTransforms,
|
|
14
|
+
createGenerationContext,
|
|
15
|
+
runGenerationAbort,
|
|
16
|
+
runGenerationError,
|
|
17
|
+
runGenerationFinish,
|
|
18
|
+
runGenerationStart,
|
|
19
|
+
runGenerationUsage,
|
|
20
|
+
} from '../middleware/run'
|
|
21
|
+
import {
|
|
22
|
+
abortReasonMessage,
|
|
23
|
+
createActivityAbortControls,
|
|
24
|
+
isActivityAbortError,
|
|
25
|
+
raceWithAbort,
|
|
26
|
+
} from '../../utilities/activity-abort'
|
|
27
|
+
import type { InternalLogger } from '../../logger/internal-logger'
|
|
28
|
+
import type { DebugOption } from '../../logger/types'
|
|
29
|
+
import type { GenerationMiddleware } from '../middleware/types'
|
|
30
|
+
import type { VoiceAdapter } from './adapter'
|
|
31
|
+
import type { StreamChunk, VoiceResult } from '../../types'
|
|
32
|
+
|
|
33
|
+
// ===========================
|
|
34
|
+
// Activity Kind
|
|
35
|
+
// ===========================
|
|
36
|
+
|
|
37
|
+
/** The adapter kind this activity handles */
|
|
38
|
+
export const kind = 'voice' as const
|
|
39
|
+
|
|
40
|
+
// ===========================
|
|
41
|
+
// Type Extraction Helpers
|
|
42
|
+
// ===========================
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Extract provider options from a VoiceAdapter via ~types.
|
|
46
|
+
*/
|
|
47
|
+
export type VoiceProviderOptions<TAdapter> = TAdapter extends {
|
|
48
|
+
'~types': { providerOptions: infer P extends object }
|
|
49
|
+
}
|
|
50
|
+
? P
|
|
51
|
+
: object
|
|
52
|
+
|
|
53
|
+
// ===========================
|
|
54
|
+
// Activity Options Type
|
|
55
|
+
// ===========================
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Options for the voice activity.
|
|
59
|
+
* The model is extracted from the adapter's model property.
|
|
60
|
+
*
|
|
61
|
+
* @template TAdapter - The voice adapter type
|
|
62
|
+
* @template TStream - Whether to stream the output
|
|
63
|
+
*/
|
|
64
|
+
export interface VoiceActivityOptions<
|
|
65
|
+
TAdapter extends VoiceAdapter<string, VoiceProviderOptions<TAdapter>>,
|
|
66
|
+
TStream extends boolean = false,
|
|
67
|
+
> {
|
|
68
|
+
/** The voice adapter to use (must be created with a model) */
|
|
69
|
+
adapter: TAdapter & { kind: typeof kind }
|
|
70
|
+
/**
|
|
71
|
+
* Text description of the voice to create, for design-capable models
|
|
72
|
+
* (e.g. `'A warm, gravelly narrator in his sixties'`).
|
|
73
|
+
*/
|
|
74
|
+
prompt?: string
|
|
75
|
+
/**
|
|
76
|
+
* Reference audio of the speaker to clone, for clone-capable models.
|
|
77
|
+
* Accepts a base64 string, base64 data URL, File, Blob, or ArrayBuffer.
|
|
78
|
+
* Remote URLs are not accepted; read the file and pass the bytes.
|
|
79
|
+
*/
|
|
80
|
+
referenceAudio?: string | File | Blob | ArrayBuffer
|
|
81
|
+
/**
|
|
82
|
+
* Name to store the voice under in the provider's voice library. Check
|
|
83
|
+
* `saved` on each returned voice to see whether it was actually persisted.
|
|
84
|
+
*/
|
|
85
|
+
name?: string
|
|
86
|
+
/** Human-readable description stored alongside the voice */
|
|
87
|
+
description?: string
|
|
88
|
+
/** Provider-specific options for voice creation */
|
|
89
|
+
modelOptions?: VoiceProviderOptions<TAdapter>
|
|
90
|
+
/**
|
|
91
|
+
* Whether to stream the generation result.
|
|
92
|
+
* When true, returns an AsyncIterable<StreamChunk> for streaming transport.
|
|
93
|
+
* When false or not provided, returns a Promise<VoiceResult>.
|
|
94
|
+
*
|
|
95
|
+
* @default false
|
|
96
|
+
*/
|
|
97
|
+
stream?: TStream
|
|
98
|
+
/**
|
|
99
|
+
* Enable debug logging. Pass `true` to enable all categories, `false` to
|
|
100
|
+
* silence everything including errors, or a `DebugConfig` object for granular
|
|
101
|
+
* control and/or a custom `Logger`.
|
|
102
|
+
*/
|
|
103
|
+
debug?: DebugOption
|
|
104
|
+
/**
|
|
105
|
+
* Observe-only middleware notified on start, usage, success, and error. Pass
|
|
106
|
+
* `otelMiddleware()` to emit OpenTelemetry spans, or implement the
|
|
107
|
+
* `GenerationMiddleware` contract for a custom backend.
|
|
108
|
+
*/
|
|
109
|
+
middleware?: Array<GenerationMiddleware>
|
|
110
|
+
/** Stable conversation/thread id for correlating this run when persisted. */
|
|
111
|
+
threadId?: string
|
|
112
|
+
/** Stable run id for correlating this run when persisted. */
|
|
113
|
+
runId?: string
|
|
114
|
+
/**
|
|
115
|
+
* Maximum duration of this activity invocation in milliseconds.
|
|
116
|
+
* No SDK-wide default — choose a value suitable for the provider and job.
|
|
117
|
+
* Composed with {@link abortSignal}; the first abort wins.
|
|
118
|
+
*/
|
|
119
|
+
timeout?: number
|
|
120
|
+
/**
|
|
121
|
+
* Caller cancellation signal (request disconnects, job/runtime cancellation).
|
|
122
|
+
* Composed with {@link timeout} into an effective signal forwarded to the
|
|
123
|
+
* adapter. Request-specific — not stored on global provider client config.
|
|
124
|
+
*/
|
|
125
|
+
abortSignal?: AbortSignal
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// ===========================
|
|
129
|
+
// Activity Result Type
|
|
130
|
+
// ===========================
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Result type for the voice activity.
|
|
134
|
+
* - If stream is true: AsyncIterable<StreamChunk>
|
|
135
|
+
* - Otherwise: Promise<VoiceResult>
|
|
136
|
+
*/
|
|
137
|
+
export type VoiceActivityResult<TStream extends boolean = false> =
|
|
138
|
+
TStream extends true ? AsyncIterable<StreamChunk> : Promise<VoiceResult>
|
|
139
|
+
|
|
140
|
+
function createId(prefix: string): string {
|
|
141
|
+
return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// ===========================
|
|
145
|
+
// Activity Implementation
|
|
146
|
+
// ===========================
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Voice activity - creates a reusable voice.
|
|
150
|
+
*
|
|
151
|
+
* Providers create voices in one of two ways, and some support both: design a
|
|
152
|
+
* new voice from a text description, or clone one from reference audio. Either
|
|
153
|
+
* way the result carries voice ids you pass back to `generateSpeech()`.
|
|
154
|
+
*
|
|
155
|
+
* @example Design a voice from a description
|
|
156
|
+
* ```ts
|
|
157
|
+
* import { generateVoice, generateSpeech } from '@tanstack/ai'
|
|
158
|
+
* import { elevenlabsVoiceDesign, elevenlabsSpeech } from '@tanstack/ai-elevenlabs'
|
|
159
|
+
*
|
|
160
|
+
* const designed = await generateVoice({
|
|
161
|
+
* adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
|
|
162
|
+
* prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',
|
|
163
|
+
* })
|
|
164
|
+
*
|
|
165
|
+
* const [preview] = designed.voices
|
|
166
|
+
* if (!preview) throw new Error('No voice candidates returned')
|
|
167
|
+
*
|
|
168
|
+
* const speech = await generateSpeech({
|
|
169
|
+
* adapter: elevenlabsSpeech('eleven_v3'),
|
|
170
|
+
* text: 'Once upon a time...',
|
|
171
|
+
* voice: preview.voiceId,
|
|
172
|
+
* })
|
|
173
|
+
* ```
|
|
174
|
+
*
|
|
175
|
+
* @example Save the voice to the provider's library
|
|
176
|
+
* ```ts
|
|
177
|
+
* const saved = await generateVoice({
|
|
178
|
+
* adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
|
|
179
|
+
* prompt: 'A bright, upbeat product demo host',
|
|
180
|
+
* name: 'Demo Host',
|
|
181
|
+
* description: 'Bright, upbeat, mid-30s',
|
|
182
|
+
* })
|
|
183
|
+
* ```
|
|
184
|
+
*/
|
|
185
|
+
export function generateVoice<
|
|
186
|
+
TAdapter extends VoiceAdapter<string, VoiceProviderOptions<TAdapter>>,
|
|
187
|
+
TStream extends boolean = false,
|
|
188
|
+
>(
|
|
189
|
+
options: VoiceActivityOptions<TAdapter, TStream>,
|
|
190
|
+
): VoiceActivityResult<TStream> {
|
|
191
|
+
if (options.stream) {
|
|
192
|
+
return streamGenerationResult(
|
|
193
|
+
// Only `runId` is taken from the resolved wire identity — see the
|
|
194
|
+
// matching note in `generateSpeech`.
|
|
195
|
+
(resolved) => runGenerateVoice({ ...options, runId: resolved.runId }),
|
|
196
|
+
options,
|
|
197
|
+
) as VoiceActivityResult<TStream>
|
|
198
|
+
}
|
|
199
|
+
return runGenerateVoice(options) as VoiceActivityResult<TStream>
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Run the core voice generation logic (non-streaming).
|
|
204
|
+
*/
|
|
205
|
+
async function runGenerateVoice<
|
|
206
|
+
TAdapter extends VoiceAdapter<string, VoiceProviderOptions<TAdapter>>,
|
|
207
|
+
>(options: VoiceActivityOptions<TAdapter, boolean>): Promise<VoiceResult> {
|
|
208
|
+
const {
|
|
209
|
+
adapter,
|
|
210
|
+
stream: _stream,
|
|
211
|
+
debug: _debug,
|
|
212
|
+
middleware,
|
|
213
|
+
threadId,
|
|
214
|
+
runId,
|
|
215
|
+
timeout,
|
|
216
|
+
abortSignal: callerAbortSignal,
|
|
217
|
+
...rest
|
|
218
|
+
} = options
|
|
219
|
+
|
|
220
|
+
if (rest.prompt == null && rest.referenceAudio == null) {
|
|
221
|
+
throw new Error(
|
|
222
|
+
'generateVoice() requires `prompt` (design a new voice) or `referenceAudio` (clone an existing one).',
|
|
223
|
+
)
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
const model = adapter.model
|
|
227
|
+
const requestId = createId('voice')
|
|
228
|
+
const startTime = Date.now()
|
|
229
|
+
const logger: InternalLogger = resolveDebugOption(options.debug)
|
|
230
|
+
const abortControls = createActivityAbortControls({
|
|
231
|
+
timeout,
|
|
232
|
+
abortSignal: callerAbortSignal,
|
|
233
|
+
})
|
|
234
|
+
|
|
235
|
+
const mwCtx = createGenerationContext({
|
|
236
|
+
requestId,
|
|
237
|
+
activity: 'voice',
|
|
238
|
+
provider: adapter.name,
|
|
239
|
+
model,
|
|
240
|
+
modelOptions: rest.modelOptions,
|
|
241
|
+
artifactInputs: {
|
|
242
|
+
prompt: rest.prompt,
|
|
243
|
+
name: rest.name,
|
|
244
|
+
description: rest.description,
|
|
245
|
+
},
|
|
246
|
+
threadId,
|
|
247
|
+
runId,
|
|
248
|
+
createId,
|
|
249
|
+
})
|
|
250
|
+
|
|
251
|
+
await runGenerationStart(middleware, mwCtx)
|
|
252
|
+
|
|
253
|
+
aiEventClient.emit('voice:request:started', {
|
|
254
|
+
requestId,
|
|
255
|
+
provider: adapter.name,
|
|
256
|
+
model,
|
|
257
|
+
prompt: rest.prompt,
|
|
258
|
+
name: rest.name,
|
|
259
|
+
description: rest.description,
|
|
260
|
+
hasReferenceAudio: rest.referenceAudio != null,
|
|
261
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
262
|
+
timestamp: startTime,
|
|
263
|
+
})
|
|
264
|
+
|
|
265
|
+
logger.request(`activity=generateVoice provider=${adapter.name}`, {
|
|
266
|
+
provider: adapter.name,
|
|
267
|
+
model,
|
|
268
|
+
})
|
|
269
|
+
|
|
270
|
+
try {
|
|
271
|
+
const rawResult = await raceWithAbort(
|
|
272
|
+
adapter.generateVoice({
|
|
273
|
+
...rest,
|
|
274
|
+
model,
|
|
275
|
+
logger,
|
|
276
|
+
...(abortControls.signal ? { abortSignal: abortControls.signal } : {}),
|
|
277
|
+
}),
|
|
278
|
+
abortControls.signal,
|
|
279
|
+
)
|
|
280
|
+
abortControls.clear()
|
|
281
|
+
const result = await applyGenerationResultTransforms(mwCtx, rawResult)
|
|
282
|
+
const duration = Date.now() - startTime
|
|
283
|
+
|
|
284
|
+
aiEventClient.emit('voice:request:completed', {
|
|
285
|
+
requestId,
|
|
286
|
+
provider: adapter.name,
|
|
287
|
+
model,
|
|
288
|
+
voiceIds: result.voices.map((voice) => voice.voiceId),
|
|
289
|
+
voiceCount: result.voices.length,
|
|
290
|
+
previewText: result.previewText,
|
|
291
|
+
duration,
|
|
292
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
293
|
+
timestamp: Date.now(),
|
|
294
|
+
})
|
|
295
|
+
|
|
296
|
+
if (result.usage) {
|
|
297
|
+
aiEventClient.emit('voice:usage', {
|
|
298
|
+
requestId,
|
|
299
|
+
model,
|
|
300
|
+
usage: result.usage,
|
|
301
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
302
|
+
timestamp: Date.now(),
|
|
303
|
+
})
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
logger.output(`activity=generateVoice voices=${result.voices.length}`, {
|
|
307
|
+
voices: result.voices.length,
|
|
308
|
+
})
|
|
309
|
+
|
|
310
|
+
if (result.usage) await runGenerationUsage(middleware, mwCtx, result.usage)
|
|
311
|
+
await runGenerationFinish(middleware, mwCtx, {
|
|
312
|
+
duration,
|
|
313
|
+
usage: result.usage,
|
|
314
|
+
})
|
|
315
|
+
|
|
316
|
+
return result
|
|
317
|
+
} catch (error) {
|
|
318
|
+
abortControls.clear()
|
|
319
|
+
const duration = Date.now() - startTime
|
|
320
|
+
const err = error as Error
|
|
321
|
+
aiEventClient.emit('voice:request:error', {
|
|
322
|
+
requestId,
|
|
323
|
+
provider: adapter.name,
|
|
324
|
+
model,
|
|
325
|
+
error: { message: err.message, name: err.name },
|
|
326
|
+
duration,
|
|
327
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
328
|
+
timestamp: Date.now(),
|
|
329
|
+
})
|
|
330
|
+
if (isActivityAbortError(error, abortControls.signal)) {
|
|
331
|
+
await runGenerationAbort(middleware, mwCtx, {
|
|
332
|
+
reason: abortReasonMessage(error, abortControls.signal),
|
|
333
|
+
duration,
|
|
334
|
+
})
|
|
335
|
+
} else {
|
|
336
|
+
await runGenerationError(middleware, mwCtx, {
|
|
337
|
+
error,
|
|
338
|
+
duration,
|
|
339
|
+
})
|
|
340
|
+
}
|
|
341
|
+
logger.errors('generateVoice activity failed', {
|
|
342
|
+
error,
|
|
343
|
+
source: 'generateVoice',
|
|
344
|
+
})
|
|
345
|
+
throw error
|
|
346
|
+
}
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
// ===========================
|
|
350
|
+
// Options Factory
|
|
351
|
+
// ===========================
|
|
352
|
+
|
|
353
|
+
/**
|
|
354
|
+
* Create typed options for the generateVoice() function without executing.
|
|
355
|
+
*/
|
|
356
|
+
export function createVoiceOptions<
|
|
357
|
+
TAdapter extends VoiceAdapter<string, VoiceProviderOptions<TAdapter>>,
|
|
358
|
+
TStream extends boolean = false,
|
|
359
|
+
>(
|
|
360
|
+
options: VoiceActivityOptions<TAdapter, TStream>,
|
|
361
|
+
): VoiceActivityOptions<TAdapter, TStream> {
|
|
362
|
+
return options
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
// Re-export adapter types
|
|
366
|
+
export type {
|
|
367
|
+
VoiceAdapter,
|
|
368
|
+
VoiceAdapterConfig,
|
|
369
|
+
AnyVoiceAdapter,
|
|
370
|
+
} from './adapter'
|
|
371
|
+
export { BaseVoiceAdapter } from './adapter'
|