@tanstack/ai 0.55.0 → 0.57.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/README.md +42 -16
  2. package/dist/esm/activities/chat/tools/tool-calls.js +1 -0
  3. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  4. package/dist/esm/activities/evaluate/adapter.d.ts +160 -0
  5. package/dist/esm/activities/evaluate/adapter.js +23 -0
  6. package/dist/esm/activities/evaluate/adapter.js.map +1 -0
  7. package/dist/esm/activities/evaluate/index.d.ts +255 -0
  8. package/dist/esm/activities/evaluate/index.js +317 -0
  9. package/dist/esm/activities/evaluate/index.js.map +1 -0
  10. package/dist/esm/activities/generateSpeech/adapter.d.ts +39 -1
  11. package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
  12. package/dist/esm/activities/generateSpeech/index.d.ts +55 -5
  13. package/dist/esm/activities/generateSpeech/index.js +53 -3
  14. package/dist/esm/activities/generateSpeech/index.js.map +1 -1
  15. package/dist/esm/activities/generateVoice/adapter.d.ts +62 -0
  16. package/dist/esm/activities/generateVoice/adapter.js +23 -0
  17. package/dist/esm/activities/generateVoice/adapter.js.map +1 -0
  18. package/dist/esm/activities/generateVoice/index.d.ts +133 -0
  19. package/dist/esm/activities/generateVoice/index.js +184 -0
  20. package/dist/esm/activities/generateVoice/index.js.map +1 -0
  21. package/dist/esm/activities/index.d.ts +10 -4
  22. package/dist/esm/activities/index.js +14 -10
  23. package/dist/esm/activities/middleware/types.d.ts +1 -1
  24. package/dist/esm/client.d.ts +3 -2
  25. package/dist/esm/client.js +21 -3
  26. package/dist/esm/client.js.map +1 -1
  27. package/dist/esm/index.d.ts +4 -2
  28. package/dist/esm/index.js +5 -2
  29. package/dist/esm/middlewares/otel.js +2 -0
  30. package/dist/esm/middlewares/otel.js.map +1 -1
  31. package/dist/esm/realtime/index.d.ts +1 -1
  32. package/dist/esm/realtime/index.js +1 -1
  33. package/dist/esm/realtime/index.js.map +1 -1
  34. package/dist/esm/types.d.ts +225 -2
  35. package/package.json +3 -3
  36. package/skills/ai-core/media-generation/SKILL.md +132 -6
  37. package/src/activities/chat/tools/tool-calls.ts +9 -0
  38. package/src/activities/evaluate/adapter.ts +212 -0
  39. package/src/activities/evaluate/index.ts +614 -0
  40. package/src/activities/generateSpeech/adapter.ts +47 -1
  41. package/src/activities/generateSpeech/index.ts +149 -8
  42. package/src/activities/generateVoice/adapter.ts +89 -0
  43. package/src/activities/generateVoice/index.ts +371 -0
  44. package/src/activities/index.ts +69 -0
  45. package/src/activities/middleware/types.ts +2 -0
  46. package/src/client.ts +35 -8
  47. package/src/index.ts +21 -0
  48. package/src/middlewares/otel.ts +2 -0
  49. package/src/realtime/index.ts +1 -1
  50. package/src/types.ts +246 -2
@@ -6,7 +6,9 @@ description: >
6
6
  generateAudio() with geminiAudio/falAudio, generateVideo() with async
7
7
  polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
8
8
  per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech/elevenlabsSpeech,
9
- generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
9
+ generateTranscription() with openaiTranscription/byteplusTranscription,
10
+ generateVoice() with elevenlabsVoiceDesign (create a voice, then speak with it).
11
+ React hooks:
10
12
  useGenerateImage, useGenerateAudio,
11
13
  useGenerateSpeech, useTranscription, useGenerateVideo.
12
14
  TanStack Start server function integration with toServerSentEventsResponse.
@@ -21,6 +23,7 @@ sources:
21
23
  - 'TanStack/ai:docs/media/video-generation.md'
22
24
  - 'TanStack/ai:docs/media/text-to-speech.md'
23
25
  - 'TanStack/ai:docs/adapters/elevenlabs.md'
26
+ - 'TanStack/ai:docs/media/voice-creation.md'
24
27
  - 'TanStack/ai:docs/media/transcription.md'
25
28
  - 'TanStack/ai:docs/advanced/debug-logging.md'
26
29
  ---
@@ -406,7 +409,126 @@ const { generate, result, isLoading } = useGenerateSpeech({
406
409
  // Play: <audio src={`data:audio/${result.format};base64,${result.audio}`} controls />
407
410
  ```
408
411
 
409
- ### 4. Audio Transcription
412
+ **Dialogue (`turns`) and timings (`timestamps`).** `text` + `voice` is one
413
+ speaker. For a multi-voice script pass `turns` instead of `text` (they are
414
+ mutually exclusive), and set `timestamps: true` to get `result.alignment`
415
+ (per character or per word, `alignment.unit` says which) and `result.segments`
416
+ (one per turn or per sentence). All times are seconds.
417
+
418
+ ```typescript
419
+ import { generateSpeech } from '@tanstack/ai'
420
+ import { byteplusSpeech } from '@tanstack/ai-byteplus'
421
+
422
+ // Second voice id comes from the BytePlus voice list.
423
+ const SECOND_VOICE = 'your-second-voice-id'
424
+
425
+ const result = await generateSpeech({
426
+ adapter: byteplusSpeech('seed-audio-1.0'),
427
+ turns: [
428
+ { text: 'Do you sell picks?', voice: 'en_female_stokie_uranus_bigtts' },
429
+ { text: 'By the till.', voice: SECOND_VOICE },
430
+ ],
431
+ timestamps: true,
432
+ })
433
+
434
+ result.alignment?.endSeconds.at(-1) // where speech stops, not where the file does
435
+ result.segments?.[0] // { startSeconds, endSeconds, turnIndex?, voice?, text? }
436
+ ```
437
+
438
+ Both are adapter capabilities, not universal. The activity rejects the request
439
+ before it reaches the provider when the adapter cannot do it, so read
440
+ `adapter.capabilities` rather than guessing:
441
+
442
+ | Adapter | `maxSpeakers` | `timestamps` |
443
+ | ----------------------- | ------------- | ------------------------------------------------ |
444
+ | `byteplusSpeech` | 3 | yes (`enable_subtitle`, word + sentence) |
445
+ | `elevenlabsSpeech` | 10 | yes (character, plus voice segments on dialogue) |
446
+ | `geminiSpeech` | 2 | no |
447
+ | every other TTS adapter | not supported | no |
448
+
449
+ ### 4. Voice Creation
450
+
451
+ Adapter: `elevenlabsVoiceDesign` (`eleven_ttv_v3`, `eleven_multilingual_ttv_v2`).
452
+
453
+ `generateVoice()` makes a voice that does not exist in any catalog, either
454
+ from a text description or from a clip of a real speaker. It returns voice ids
455
+ you pass straight back to `generateSpeech()` as `voice`.
456
+
457
+ > Pass `prompt`, or `referenceAudio`, or both — the activity throws when
458
+ > neither is given. ElevenLabs always needs `prompt`, because its design
459
+ > endpoint requires a description, and only `eleven_ttv_v3` accepts
460
+ > `referenceAudio`. Without a `name` you get **previews**, which expire;
461
+ > with a `name` the best candidate is kept in the provider's voice library.
462
+ > Check `saved` on each returned voice rather than assuming. Remote audio
463
+ > URLs are rejected: read the file and pass bytes.
464
+
465
+ ```typescript
466
+ import { generateSpeech, generateVoice } from '@tanstack/ai'
467
+ import {
468
+ elevenlabsSpeech,
469
+ elevenlabsVoiceDesign,
470
+ } from '@tanstack/ai-elevenlabs'
471
+
472
+ const designed = await generateVoice({
473
+ adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
474
+ prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',
475
+ name: 'Irish Narrator', // omit to audition previews instead
476
+ })
477
+
478
+ const [voice] = designed.voices
479
+ if (!voice) throw new Error('The provider returned no voices.')
480
+
481
+ // voice.voiceId -> pass to generateSpeech()
482
+ // voice.audio -> base64 preview, when the provider returns one
483
+ // voice.saved -> true only when it is in the provider's library
484
+ // voice.status -> 'ready' on every adapter today
485
+
486
+ const speech = await generateSpeech({
487
+ adapter: elevenlabsSpeech('eleven_v3'),
488
+ text: 'Once upon a time...',
489
+ voice: voice.voiceId,
490
+ })
491
+ ```
492
+
493
+ **Status.** Every returned voice carries `status`. It is `'ready'` on every
494
+ adapter today, because they all finish the voice before returning. The
495
+ `'training'` and `'failed'` members exist for providers that build a voice
496
+ asynchronously; no adapter returns them yet, so do not write polling code
497
+ against them.
498
+
499
+ **Finding voices again.** `generateVoice()` hands back an id you are expected
500
+ to store. `listVoices({ adapter: <a TTS adapter>, origins })` reads the
501
+ account catalog back when you did not.
502
+
503
+ ```typescript
504
+ import { listVoices } from '@tanstack/ai'
505
+ import { elevenlabsSpeech } from '@tanstack/ai-elevenlabs'
506
+
507
+ const { voices } = await listVoices({
508
+ adapter: elevenlabsSpeech('eleven_v3'),
509
+ origins: ['generated', 'cloned'],
510
+ })
511
+ ```
512
+
513
+ `listVoices` hangs off the **TTS** adapter, not the voice adapter, because
514
+ `voice` is a `generateSpeech()` option — that is where the id gets consumed.
515
+ It is OPTIONAL, and only providers with a per-account catalog implement it.
516
+ Where the catalog is fixed the package publishes it instead — `GeminiTTSVoices`
517
+ from `@tanstack/ai-gemini`, or the `OpenAITTSVoice` union from
518
+ `@tanstack/ai-openai`. Prefer those: a type union beats a network call.
519
+ Calling `listVoices()` on such an adapter throws and points at them.
520
+
521
+ There is no React hook for this activity. Call it from a server route or
522
+ server function and return the result as JSON.
523
+
524
+ `elevenlabsVoiceDesign` is the only `generateVoice()` adapter in this repo.
525
+ xAI, BytePlus, and fal.ai each publish a voice-cloning API and are the
526
+ candidates for the next one, but none is implemented — do not write code
527
+ against them from this file.
528
+
529
+ OpenAI, Gemini, and Cloudflare have fixed voice catalogs and will not get one.
530
+
531
+ ### 5. Audio Transcription
410
532
 
411
533
  Adapters: `openaiTranscription` (whisper-1, gpt-4o-transcribe,
412
534
  gpt-4o-mini-transcribe, gpt-4o-transcribe-diarize) and `byteplusTranscription`
@@ -486,7 +608,7 @@ const { generate, result, isLoading } = useTranscription({
486
608
  // Trigger: generate({ audio: dataUrl, language: 'en' })
487
609
  ```
488
610
 
489
- ### 5. Video Generation (Experimental -- async polling)
611
+ ### 6. Video Generation (Experimental -- async polling)
490
612
 
491
613
  Video generation uses a jobs/polling architecture. The server creates a job,
492
614
  polls for status, and streams updates to the client. Adapters: `openaiVideo`
@@ -662,7 +784,7 @@ const { generate, result, jobId, videoStatus, isLoading } = useGenerateVideo({
662
784
  // result (on completion): { url }
663
785
  ```
664
786
 
665
- ### 6. Cost tracking (fal billable units)
787
+ ### 7. Cost tracking (fal billable units)
666
788
 
667
789
  fal bills media generation by usage-based units, not tokens. Every fal media
668
790
  adapter (`falImage`, `falAudio`, `falSpeech`, `falTranscription`, `falVideo`)
@@ -692,7 +814,7 @@ if (result.usage?.billed) {
692
814
  For video, the units arrive with the completed result: `getVideoJobStatus()`
693
815
  returns `usage` and emits a `video:usage` devtools event when fal reports it.
694
816
 
695
- ### 7. Durable persistence (job lifecycle + artifact bytes)
817
+ ### 8. Durable persistence (job lifecycle + artifact bytes)
696
818
 
697
819
  To make generations survive a server restart and be re-served later, add
698
820
  `withGenerationPersistence` from `@tanstack/ai-persistence` as generation
@@ -1014,7 +1136,11 @@ generateAudio({
1014
1136
 
1015
1137
  ### g. MEDIUM: Gemini TTS multi-speaker with 0 or 3+ speakers
1016
1138
 
1017
- `multiSpeakerVoiceConfig.speakerVoiceConfigs` is validated to be length 1 or 2. Passing an empty array or three+ entries throws at the adapter boundary
1139
+ Prefer `turns` for new code: it builds `multiSpeakerVoiceConfig` and the
1140
+ labelled prompt for you, and the two-speaker cap is enforced by the activity
1141
+ from `capabilities.maxSpeakers`.
1142
+
1143
+ The hand-rolled form below still works. `multiSpeakerVoiceConfig.speakerVoiceConfigs` is validated to be length 1 or 2. Passing an empty array or three+ entries throws at the adapter boundary
1018
1144
  (not at Gemini's API) with a clear error. Don't try to work around it with
1019
1145
  `as any`.
1020
1146
 
@@ -233,6 +233,15 @@ export class ToolCallManager<
233
233
  * Add a TOOL_CALL_START event to begin tracking a tool call (AG-UI)
234
234
  */
235
235
  addToolCallStartEvent(event: ToolCallStartEvent): void {
236
+ // AG-UI's TOOL_CALL_START carries no index, and a non-first-party or
237
+ // malformed producer can send a second START for a toolCallId that is
238
+ // already tracked. Without this guard, a repeat with the same index
239
+ // overwrites the slot (wiping any TOOL_CALL_ARGS already accumulated),
240
+ // and a repeat with a missing/different index inserts a duplicate row
241
+ // that getToolCalls() returns twice, running the tool twice.
242
+ for (const toolCall of this.toolCallsMap.values()) {
243
+ if (toolCall.id === event.toolCallId) return
244
+ }
236
245
  const index = (event as AdapterYieldChunk).index ?? this.toolCallsMap.size
237
246
  const name = event.toolCallName ?? event.toolName
238
247
  this.toolCallsMap.set(index, {
@@ -0,0 +1,212 @@
1
+ import type { InternalLogger } from '../../logger/internal-logger'
2
+ import type { TokenUsage } from '../../types'
3
+
4
+ /**
5
+ * Configuration for evaluate adapter instances.
6
+ */
7
+ export interface EvaluateAdapterConfig {
8
+ apiKey?: string
9
+ baseUrl?: string
10
+ timeout?: number
11
+ headers?: Record<string, string>
12
+ }
13
+
14
+ /**
15
+ * Shared JSON value for `state` and question `instructions`.
16
+ * A JSON array is one value, not a batch.
17
+ */
18
+ export type EvaluateJsonValue = string | object | Array<unknown>
19
+
20
+ /** Content the model judges. A JSON array is one state, not a batch. */
21
+ export type EvaluateState = EvaluateJsonValue
22
+
23
+ /** Question text. Matches TypeSafe: string, object, or array. */
24
+ export type EvaluateInstructions = EvaluateJsonValue
25
+
26
+ /**
27
+ * TypeSafe choice question on the adapter wire.
28
+ *
29
+ * Generic parameters:
30
+ * - TOptions: option key to description (or `null` when the key is enough)
31
+ */
32
+ export interface WireChoiceQuestion<
33
+ TOptions extends Record<string, string | null> = Record<
34
+ string,
35
+ string | null
36
+ >,
37
+ > {
38
+ type: 'choice'
39
+ instructions: EvaluateInstructions
40
+ criteria: TOptions
41
+ }
42
+
43
+ /**
44
+ * TypeSafe score question on the adapter wire.
45
+ *
46
+ * Generic parameters:
47
+ * - TLevels: ordered level labels, at least two
48
+ */
49
+ export interface WireScoreQuestion<
50
+ TLevels extends ReadonlyArray<string> = ReadonlyArray<string>,
51
+ > {
52
+ type: 'score'
53
+ instructions: EvaluateInstructions
54
+ criteria: TLevels
55
+ }
56
+
57
+ /**
58
+ * TypeSafe yes/no question on the adapter wire.
59
+ * Public helpers call this `boolean`. The wire type is `noul`.
60
+ */
61
+ export interface WireNoulQuestion {
62
+ type: 'noul'
63
+ instructions: EvaluateInstructions
64
+ criteria?: {
65
+ true?: string
66
+ false?: string
67
+ }
68
+ }
69
+
70
+ /** Question payload adapters send to the provider. */
71
+ export type WireQuestion =
72
+ | WireChoiceQuestion
73
+ | WireScoreQuestion
74
+ | WireNoulQuestion
75
+
76
+ /** TypeSafe choice answer. Adapters do not invent a public `.value`. */
77
+ export interface WireChoiceAnswer {
78
+ type: 'choice'
79
+ choice: string
80
+ probabilities: Record<string, number>
81
+ confidence: number
82
+ }
83
+
84
+ /** TypeSafe score answer. `score` is the raw fraction. */
85
+ export interface WireScoreAnswer {
86
+ type: 'score'
87
+ score: number
88
+ legend: Record<string, string>
89
+ probabilities: Record<string, number>
90
+ confidence: number
91
+ }
92
+
93
+ /** TypeSafe yes/no answer. `noul` is P(true). */
94
+ export interface WireNoulAnswer {
95
+ type: 'noul'
96
+ noul: number
97
+ }
98
+
99
+ /** Provider payload for one question. The activity maps this to a unified answer. */
100
+ export type WireAnswer = WireChoiceAnswer | WireScoreAnswer | WireNoulAnswer
101
+
102
+ /**
103
+ * Options passed to {@link EvaluateAdapter.evaluate}.
104
+ */
105
+ export interface EvaluateOptions<
106
+ TProviderOptions extends object = Record<string, unknown>,
107
+ > {
108
+ model: string
109
+ /** Shared state every question judges. A JSON array is one state, not a batch. */
110
+ state: EvaluateState
111
+ /** TypeSafe wire questions, keyed by the caller's question ids. */
112
+ questions: Record<string, WireQuestion>
113
+ /** Provider-specific options forwarded by `decide()`. */
114
+ modelOptions?: TProviderOptions
115
+ /** Forwarded to the provider request for cancellation. */
116
+ abortSignal?: AbortSignal
117
+ /**
118
+ * Internal logger threaded from `decide()`. Adapters must call
119
+ * `logger.request()` before the provider call and `logger.errors()` in catch
120
+ * blocks.
121
+ */
122
+ logger: InternalLogger
123
+ }
124
+
125
+ /**
126
+ * Provider-level evaluate result. Adapters return the wire payload plus usage.
127
+ * The activity maps answers to the unified public shape.
128
+ */
129
+ export interface EvaluateAdapterResult {
130
+ /** Resolved model id from the provider. */
131
+ model: string
132
+ answers: Record<string, WireAnswer>
133
+ usage: TokenUsage
134
+ }
135
+
136
+ /**
137
+ * Evaluate adapter interface with pre-resolved generics.
138
+ *
139
+ * An adapter is created by a provider function: `provider('model')` → `adapter`.
140
+ * All type resolution happens at the provider call site, not in this interface.
141
+ *
142
+ * Generic parameters:
143
+ * - TModel: The specific model name (e.g. `'jev-latest'`)
144
+ * - TProviderOptions: Provider-specific options (already resolved)
145
+ */
146
+ export interface EvaluateAdapter<
147
+ TModel extends string = string,
148
+ TProviderOptions extends object = Record<string, unknown>,
149
+ > {
150
+ /** Discriminator for adapter kind */
151
+ readonly kind: 'evaluate'
152
+ /** Adapter name identifier */
153
+ readonly name: string
154
+ /** The model this adapter is configured for */
155
+ readonly model: TModel
156
+
157
+ /**
158
+ * @internal Type-only properties for inference. Not assigned at runtime.
159
+ */
160
+ '~types': {
161
+ providerOptions: TProviderOptions
162
+ }
163
+
164
+ /**
165
+ * Evaluate typed questions against `state`. Return the provider payload.
166
+ * Do not invent unified `.value` fields. The activity maps wire answers.
167
+ */
168
+ evaluate: (
169
+ options: EvaluateOptions<TProviderOptions>,
170
+ ) => Promise<EvaluateAdapterResult>
171
+ }
172
+
173
+ /**
174
+ * An EvaluateAdapter with any/unknown type parameters.
175
+ * Useful as a constraint in generic functions and interfaces.
176
+ */
177
+ export type AnyEvaluateAdapter = EvaluateAdapter<any, any>
178
+
179
+ /**
180
+ * Abstract base class for evaluate adapters.
181
+ * Extend this class to implement an evaluate adapter for a specific provider.
182
+ *
183
+ * Generic parameters match EvaluateAdapter. The provider function resolves them.
184
+ */
185
+ export abstract class BaseEvaluateAdapter<
186
+ TModel extends string = string,
187
+ TProviderOptions extends object = Record<string, unknown>,
188
+ > implements EvaluateAdapter<TModel, TProviderOptions> {
189
+ readonly kind = 'evaluate' as const
190
+ abstract readonly name: string
191
+ readonly model: TModel
192
+
193
+ // Type-only property - never assigned at runtime
194
+ declare '~types': {
195
+ providerOptions: TProviderOptions
196
+ }
197
+
198
+ protected config: EvaluateAdapterConfig
199
+
200
+ constructor(config: EvaluateAdapterConfig = {}, model: TModel) {
201
+ this.config = config
202
+ this.model = model
203
+ }
204
+
205
+ abstract evaluate(
206
+ options: EvaluateOptions<TProviderOptions>,
207
+ ): Promise<EvaluateAdapterResult>
208
+
209
+ protected generateId(): string {
210
+ return `${this.name}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`
211
+ }
212
+ }