@tanstack/ai 0.13.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/error-payload.d.ts +12 -0
- package/dist/esm/activities/error-payload.js +25 -0
- package/dist/esm/activities/error-payload.js.map +1 -0
- package/dist/esm/activities/generateAudio/adapter.d.ts +62 -0
- package/dist/esm/activities/generateAudio/adapter.js +14 -0
- package/dist/esm/activities/generateAudio/adapter.js.map +1 -0
- package/dist/esm/activities/generateAudio/index.d.ts +74 -0
- package/dist/esm/activities/generateAudio/index.js +80 -0
- package/dist/esm/activities/generateAudio/index.js.map +1 -0
- package/dist/esm/activities/generateImage/adapter.d.ts +1 -1
- package/dist/esm/activities/generateImage/adapter.js +1 -1
- package/dist/esm/activities/generateImage/adapter.js.map +1 -1
- package/dist/esm/activities/generateSpeech/adapter.d.ts +1 -1
- package/dist/esm/activities/generateSpeech/adapter.js +1 -1
- package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +3 -3
- package/dist/esm/activities/generateSpeech/index.js +11 -0
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/activities/generateTranscription/adapter.d.ts +1 -1
- package/dist/esm/activities/generateTranscription/adapter.js +1 -1
- package/dist/esm/activities/generateTranscription/adapter.js.map +1 -1
- package/dist/esm/activities/generateTranscription/index.d.ts +3 -3
- package/dist/esm/activities/generateTranscription/index.js +11 -0
- package/dist/esm/activities/generateTranscription/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/index.js +7 -7
- package/dist/esm/activities/generateVideo/index.js.map +1 -1
- package/dist/esm/activities/index.d.ts +5 -2
- package/dist/esm/activities/index.js +11 -6
- package/dist/esm/activities/index.js.map +1 -1
- package/dist/esm/activities/stream-generation-result.js +5 -6
- package/dist/esm/activities/stream-generation-result.js.map +1 -1
- package/dist/esm/adapter-internals.d.ts +1 -0
- package/dist/esm/adapter-internals.js +3 -1
- package/dist/esm/adapter-internals.js.map +1 -1
- package/dist/esm/index.d.ts +3 -2
- package/dist/esm/index.js +3 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/stream-to-response.js +3 -8
- package/dist/esm/stream-to-response.js.map +1 -1
- package/dist/esm/types.d.ts +63 -6
- package/package.json +2 -2
- package/skills/ai-core/media-generation/SKILL.md +154 -10
- package/src/activities/error-payload.ts +35 -0
- package/src/activities/generateAudio/adapter.ts +89 -0
- package/src/activities/generateAudio/index.ts +224 -0
- package/src/activities/generateImage/adapter.ts +1 -1
- package/src/activities/generateSpeech/adapter.ts +1 -1
- package/src/activities/generateSpeech/index.ts +17 -7
- package/src/activities/generateTranscription/adapter.ts +1 -1
- package/src/activities/generateTranscription/index.ts +27 -4
- package/src/activities/generateVideo/index.ts +8 -8
- package/src/activities/index.ts +22 -0
- package/src/activities/stream-generation-result.ts +6 -7
- package/src/adapter-internals.ts +1 -0
- package/src/index.ts +4 -0
- package/src/stream-to-response.ts +5 -10
- package/src/types.ts +74 -5
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: ai-core/media-generation
|
|
3
3
|
description: >
|
|
4
|
-
Image, video, speech (TTS), and transcription generation using
|
|
4
|
+
Image, audio, video, speech (TTS), and transcription generation using
|
|
5
5
|
activity-specific adapters: generateImage() with openaiImage/geminiImage,
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
useGenerateImage,
|
|
6
|
+
generateAudio() with geminiAudio/falAudio, generateVideo() with async
|
|
7
|
+
polling, generateSpeech() with openaiSpeech, generateTranscription() with
|
|
8
|
+
openaiTranscription. React hooks: useGenerateImage, useGenerateAudio,
|
|
9
|
+
useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
9
10
|
TanStack Start server function integration with toServerSentEventsResponse.
|
|
10
11
|
type: sub-skill
|
|
11
12
|
library: tanstack-ai
|
|
@@ -14,9 +15,11 @@ sources:
|
|
|
14
15
|
- 'TanStack/ai:docs/media/generations.md'
|
|
15
16
|
- 'TanStack/ai:docs/media/generation-hooks.md'
|
|
16
17
|
- 'TanStack/ai:docs/media/image-generation.md'
|
|
18
|
+
- 'TanStack/ai:docs/media/audio-generation.md'
|
|
17
19
|
- 'TanStack/ai:docs/media/video-generation.md'
|
|
18
20
|
- 'TanStack/ai:docs/media/text-to-speech.md'
|
|
19
21
|
- 'TanStack/ai:docs/media/transcription.md'
|
|
22
|
+
- 'TanStack/ai:docs/advanced/debug-logging.md'
|
|
20
23
|
---
|
|
21
24
|
|
|
22
25
|
# Media Generation
|
|
@@ -186,7 +189,40 @@ Result shape: `ImageGenerationResult` with `images` array where each entry
|
|
|
186
189
|
has `b64Json?`, `url?`, and `revisedPrompt?`. OpenAI image URLs expire
|
|
187
190
|
after 1 hour -- download or display immediately.
|
|
188
191
|
|
|
189
|
-
### 2.
|
|
192
|
+
### 2. Audio Generation (Music, Sound Effects)
|
|
193
|
+
|
|
194
|
+
Distinct from TTS — `generateAudio()` produces non-speech audio content.
|
|
195
|
+
Supported adapters: `geminiAudio` (Lyria 3 Pro / Lyria 3 Clip) and
|
|
196
|
+
`falAudio` (MiniMax Music, DiffRhythm, Stable Audio, ElevenLabs SFX, etc.).
|
|
197
|
+
|
|
198
|
+
```typescript
|
|
199
|
+
import { generateAudio } from '@tanstack/ai'
|
|
200
|
+
import { falAudio } from '@tanstack/ai-fal'
|
|
201
|
+
|
|
202
|
+
const result = await generateAudio({
|
|
203
|
+
adapter: falAudio('fal-ai/diffrhythm'),
|
|
204
|
+
prompt: 'An upbeat electronic track with synths',
|
|
205
|
+
duration: 10,
|
|
206
|
+
})
|
|
207
|
+
|
|
208
|
+
// result.audio.url or result.audio.b64Json (provider-dependent)
|
|
209
|
+
// result.audio.contentType e.g. "audio/mpeg"
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
Client hook:
|
|
213
|
+
|
|
214
|
+
```tsx
|
|
215
|
+
import { useGenerateAudio, fetchServerSentEvents } from '@tanstack/ai-react'
|
|
216
|
+
|
|
217
|
+
const { generate, result, isLoading } = useGenerateAudio({
|
|
218
|
+
connection: fetchServerSentEvents('/api/generate/audio'),
|
|
219
|
+
})
|
|
220
|
+
|
|
221
|
+
// Trigger: generate({ prompt: 'Upbeat synths', duration: 10 })
|
|
222
|
+
// Play: <audio src={result.audio.url} controls />
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
### 3. Text-to-Speech
|
|
190
226
|
|
|
191
227
|
Adapter: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview).
|
|
192
228
|
|
|
@@ -220,7 +256,7 @@ const { generate, result, isLoading } = useGenerateSpeech({
|
|
|
220
256
|
// Play: <audio src={`data:audio/${result.format};base64,${result.audio}`} controls />
|
|
221
257
|
```
|
|
222
258
|
|
|
223
|
-
###
|
|
259
|
+
### 4. Audio Transcription
|
|
224
260
|
|
|
225
261
|
Adapter: `openaiTranscription` (whisper-1, gpt-4o-transcribe,
|
|
226
262
|
gpt-4o-mini-transcribe).
|
|
@@ -257,7 +293,7 @@ const { generate, result, isLoading } = useTranscription({
|
|
|
257
293
|
// Trigger: generate({ audio: dataUrl, language: 'en' })
|
|
258
294
|
```
|
|
259
295
|
|
|
260
|
-
###
|
|
296
|
+
### 5. Video Generation (Experimental -- async polling)
|
|
261
297
|
|
|
262
298
|
Video generation uses a jobs/polling architecture. The server creates a job,
|
|
263
299
|
polls for status, and streams updates to the client.
|
|
@@ -454,12 +490,116 @@ for (const img of result.images) {
|
|
|
454
490
|
Not all generation activities support streaming. Passing `stream: true` to
|
|
455
491
|
an activity that does not support it may hang or produce unexpected results.
|
|
456
492
|
Check the activity documentation before enabling streaming. All built-in
|
|
457
|
-
activities (`generateImage`, `
|
|
458
|
-
`generateVideo`, `summarize`) support `stream: true`,
|
|
459
|
-
`useGeneration` setups may not.
|
|
493
|
+
activities (`generateImage`, `generateAudio`, `generateSpeech`,
|
|
494
|
+
`generateTranscription`, `generateVideo`, `summarize`) support `stream: true`,
|
|
495
|
+
but custom `useGeneration` setups may not.
|
|
460
496
|
|
|
461
497
|
> Source: docs/media/generations.md.
|
|
462
498
|
|
|
499
|
+
### e. HIGH: Passing `responseMimeType` or `negativePrompt` to Gemini Lyria
|
|
500
|
+
|
|
501
|
+
Gemini's `GenerateContentConfig` (used by Lyria 3 Pro / Lyria 3 Clip) does
|
|
502
|
+
**not** support `responseMimeType` or `negativePrompt`. Lyria 3 Clip always
|
|
503
|
+
returns 30-second `audio/mp3`; Lyria 3 Pro returns `audio/mp3`. These fields
|
|
504
|
+
are not in `GeminiAudioProviderOptions` — don't reach for them via `as any`.
|
|
505
|
+
|
|
506
|
+
```typescript
|
|
507
|
+
// WRONG — both fields are silently ignored or rejected by the SDK
|
|
508
|
+
generateAudio({
|
|
509
|
+
adapter: geminiAudio('lyria-3-pro-preview'),
|
|
510
|
+
prompt: 'ambient piano',
|
|
511
|
+
modelOptions: {
|
|
512
|
+
responseMimeType: 'audio/wav', // unsupported
|
|
513
|
+
negativePrompt: 'vocals', // unsupported
|
|
514
|
+
} as any,
|
|
515
|
+
})
|
|
516
|
+
|
|
517
|
+
// CORRECT — shape the prompt itself for what you want
|
|
518
|
+
generateAudio({
|
|
519
|
+
adapter: geminiAudio('lyria-3-pro-preview'),
|
|
520
|
+
prompt: 'ambient piano, no vocals',
|
|
521
|
+
})
|
|
522
|
+
```
|
|
523
|
+
|
|
524
|
+
> Source: Gemini API `GenerateContentConfig` type; docs/media/audio-generation.md.
|
|
525
|
+
|
|
526
|
+
### f. MEDIUM: Passing `duration` to Lyria expecting it to control length
|
|
527
|
+
|
|
528
|
+
Lyria 3 Clip is fixed at 30 seconds — the `duration` option is ignored on
|
|
529
|
+
that model. Lyria 3 Pro accepts duration via natural-language in the
|
|
530
|
+
**prompt** ("2-minute ambient track with a 30-second build"), not via the
|
|
531
|
+
`duration` field. `duration` works for fal audio models (mapped to each
|
|
532
|
+
model's native field like `music_length_ms` or `seconds_total`), but not
|
|
533
|
+
for Lyria.
|
|
534
|
+
|
|
535
|
+
```typescript
|
|
536
|
+
// For Lyria: put length guidance in the prompt
|
|
537
|
+
generateAudio({
|
|
538
|
+
adapter: geminiAudio('lyria-3-pro-preview'),
|
|
539
|
+
prompt: 'A 2-minute ambient piano piece with gentle strings',
|
|
540
|
+
// duration: 120 // ← does nothing; rely on the prompt
|
|
541
|
+
})
|
|
542
|
+
|
|
543
|
+
// For fal: duration works and is translated per-model
|
|
544
|
+
generateAudio({
|
|
545
|
+
adapter: falAudio('fal-ai/minimax-music/v2'),
|
|
546
|
+
prompt: 'upbeat synth melody',
|
|
547
|
+
duration: 60, // → music_length_ms: 60_000
|
|
548
|
+
})
|
|
549
|
+
```
|
|
550
|
+
|
|
551
|
+
> Source: Google Lyria 3 docs; docs/media/audio-generation.md.
|
|
552
|
+
|
|
553
|
+
### g. MEDIUM: Gemini TTS multi-speaker with 0 or 3+ speakers
|
|
554
|
+
|
|
555
|
+
`multiSpeakerVoiceConfig.speakerVoiceConfigs` is validated to be length 1 or 2. Passing an empty array or three+ entries throws at the adapter boundary
|
|
556
|
+
(not at Gemini's API) with a clear error. Don't try to work around it with
|
|
557
|
+
`as any`.
|
|
558
|
+
|
|
559
|
+
```typescript
|
|
560
|
+
generateSpeech({
|
|
561
|
+
adapter: geminiSpeech('gemini-2.5-pro-preview-tts'),
|
|
562
|
+
text: '[Alice] Hi. [Bob] Hello!',
|
|
563
|
+
modelOptions: {
|
|
564
|
+
multiSpeakerVoiceConfig: {
|
|
565
|
+
speakerVoiceConfigs: [
|
|
566
|
+
{
|
|
567
|
+
speaker: 'Alice',
|
|
568
|
+
voiceConfig: { prebuiltVoiceConfig: { voiceName: 'Kore' } },
|
|
569
|
+
},
|
|
570
|
+
{
|
|
571
|
+
speaker: 'Bob',
|
|
572
|
+
voiceConfig: { prebuiltVoiceConfig: { voiceName: 'Puck' } },
|
|
573
|
+
},
|
|
574
|
+
],
|
|
575
|
+
},
|
|
576
|
+
},
|
|
577
|
+
})
|
|
578
|
+
```
|
|
579
|
+
|
|
580
|
+
> Source: Gemini TTS adapter validation; CodeRabbit review of PR #463.
|
|
581
|
+
|
|
582
|
+
### h. LOW: Writing a logging middleware to see media chunks flow through
|
|
583
|
+
|
|
584
|
+
Every media activity — `generateAudio`, `generateSpeech`,
|
|
585
|
+
`generateTranscription`, `generateImage`, `generateVideo` — accepts the
|
|
586
|
+
same `debug?: DebugOption` option that `chat()` does. Reach for `debug`
|
|
587
|
+
instead of wiring up logging middleware.
|
|
588
|
+
|
|
589
|
+
```typescript
|
|
590
|
+
// When a speech generation sounds wrong or a transcription returns garbage
|
|
591
|
+
generateSpeech({
|
|
592
|
+
adapter: openaiSpeech('tts-1'),
|
|
593
|
+
text: 'Hello',
|
|
594
|
+
debug: { provider: true, output: true }, // raw SDK chunks + yielded chunks
|
|
595
|
+
})
|
|
596
|
+
```
|
|
597
|
+
|
|
598
|
+
See the `ai-core/debug-logging` sub-skill for full details on categories
|
|
599
|
+
and piping into a custom logger.
|
|
600
|
+
|
|
601
|
+
> Source: docs/advanced/debug-logging.md.
|
|
602
|
+
|
|
463
603
|
---
|
|
464
604
|
|
|
465
605
|
## Cross-References
|
|
@@ -469,3 +609,7 @@ activities (`generateImage`, `generateSpeech`, `generateTranscription`,
|
|
|
469
609
|
images, `openaiSpeech` for speech, `openaiTranscription` for transcription,
|
|
470
610
|
`openaiVideo` for video). The adapter-configuration skill covers provider
|
|
471
611
|
setup, API keys, and model selection.
|
|
612
|
+
- See also: **ai-core/debug-logging/SKILL.md** -- When a media request
|
|
613
|
+
returns unexpected output or fails mid-stream, toggle `debug: true` on
|
|
614
|
+
any `generate*()` call to see request metadata, raw provider chunks, and
|
|
615
|
+
errors. Covers per-category toggling and piping into pino/winston.
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared error-narrowing helper for activities that convert thrown values
|
|
3
|
+
* into structured `RUN_ERROR` events.
|
|
4
|
+
*
|
|
5
|
+
* Accepts Error instances, objects with string-ish `message`/`code`, or bare
|
|
6
|
+
* strings; always returns a shape safe to serialize. Never leaks the full
|
|
7
|
+
* error object (which may carry request/response state from an SDK).
|
|
8
|
+
*/
|
|
9
|
+
export function toRunErrorPayload(
|
|
10
|
+
error: unknown,
|
|
11
|
+
fallbackMessage = 'Unknown error occurred',
|
|
12
|
+
): { message: string; code: string | undefined } {
|
|
13
|
+
if (error instanceof Error) {
|
|
14
|
+
const codeField = (error as Error & { code?: unknown }).code
|
|
15
|
+
return {
|
|
16
|
+
message: error.message || fallbackMessage,
|
|
17
|
+
code: typeof codeField === 'string' ? codeField : undefined,
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
if (typeof error === 'object' && error !== null) {
|
|
21
|
+
const messageField = (error as { message?: unknown }).message
|
|
22
|
+
const codeField = (error as { code?: unknown }).code
|
|
23
|
+
return {
|
|
24
|
+
message:
|
|
25
|
+
typeof messageField === 'string' && messageField.length > 0
|
|
26
|
+
? messageField
|
|
27
|
+
: fallbackMessage,
|
|
28
|
+
code: typeof codeField === 'string' ? codeField : undefined,
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
if (typeof error === 'string' && error.length > 0) {
|
|
32
|
+
return { message: error, code: undefined }
|
|
33
|
+
}
|
|
34
|
+
return { message: fallbackMessage, code: undefined }
|
|
35
|
+
}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
import type { AudioGenerationOptions, AudioGenerationResult } from '../../types'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Configuration for audio generation adapter instances
|
|
5
|
+
*/
|
|
6
|
+
export interface AudioAdapterConfig {
|
|
7
|
+
apiKey?: string
|
|
8
|
+
baseUrl?: string
|
|
9
|
+
timeout?: number
|
|
10
|
+
maxRetries?: number
|
|
11
|
+
headers?: Record<string, string>
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Audio generation adapter interface with pre-resolved generics.
|
|
16
|
+
*
|
|
17
|
+
* An adapter is created by a provider function: `provider('model')` → `adapter`
|
|
18
|
+
* All type resolution happens at the provider call site, not in this interface.
|
|
19
|
+
*
|
|
20
|
+
* Generic parameters:
|
|
21
|
+
* - TModel: The specific model name (e.g., 'fal-ai/diffrhythm')
|
|
22
|
+
* - TProviderOptions: Provider-specific options (already resolved)
|
|
23
|
+
*/
|
|
24
|
+
export interface AudioAdapter<
|
|
25
|
+
TModel extends string = string,
|
|
26
|
+
TProviderOptions extends object = Record<string, unknown>,
|
|
27
|
+
> {
|
|
28
|
+
/** Discriminator for adapter kind - used to determine API shape */
|
|
29
|
+
readonly kind: 'audio'
|
|
30
|
+
/** Adapter name identifier */
|
|
31
|
+
readonly name: string
|
|
32
|
+
/** The model this adapter is configured for */
|
|
33
|
+
readonly model: TModel
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* @internal Type-only properties for inference. Not assigned at runtime.
|
|
37
|
+
*/
|
|
38
|
+
'~types': {
|
|
39
|
+
providerOptions: TProviderOptions
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Generate audio from a text prompt
|
|
44
|
+
*/
|
|
45
|
+
generateAudio: (
|
|
46
|
+
options: AudioGenerationOptions<TProviderOptions>,
|
|
47
|
+
) => Promise<AudioGenerationResult>
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* An AudioAdapter with any/unknown type parameters.
|
|
52
|
+
* Useful as a constraint in generic functions and interfaces.
|
|
53
|
+
*/
|
|
54
|
+
export type AnyAudioAdapter = AudioAdapter<any, any>
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Abstract base class for audio generation adapters.
|
|
58
|
+
* Extend this class to implement an audio adapter for a specific provider.
|
|
59
|
+
*
|
|
60
|
+
* Generic parameters match AudioAdapter - all pre-resolved by the provider function.
|
|
61
|
+
*/
|
|
62
|
+
export abstract class BaseAudioAdapter<
|
|
63
|
+
TModel extends string = string,
|
|
64
|
+
TProviderOptions extends object = Record<string, unknown>,
|
|
65
|
+
> implements AudioAdapter<TModel, TProviderOptions> {
|
|
66
|
+
readonly kind = 'audio' as const
|
|
67
|
+
abstract readonly name: string
|
|
68
|
+
readonly model: TModel
|
|
69
|
+
|
|
70
|
+
// Type-only property - never assigned at runtime
|
|
71
|
+
declare '~types': {
|
|
72
|
+
providerOptions: TProviderOptions
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
protected config: AudioAdapterConfig
|
|
76
|
+
|
|
77
|
+
constructor(model: TModel, config: AudioAdapterConfig = {}) {
|
|
78
|
+
this.config = config
|
|
79
|
+
this.model = model
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
abstract generateAudio(
|
|
83
|
+
options: AudioGenerationOptions<TProviderOptions>,
|
|
84
|
+
): Promise<AudioGenerationResult>
|
|
85
|
+
|
|
86
|
+
protected generateId(): string {
|
|
87
|
+
return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`
|
|
88
|
+
}
|
|
89
|
+
}
|
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Audio Generation Activity
|
|
3
|
+
*
|
|
4
|
+
* Generates audio (music, sound effects, etc.) from text prompts.
|
|
5
|
+
* This is a self-contained module with implementation, types, and JSDoc.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { aiEventClient } from '@tanstack/ai-event-client'
|
|
9
|
+
import { streamGenerationResult } from '../stream-generation-result.js'
|
|
10
|
+
import { resolveDebugOption } from '../../logger/resolve'
|
|
11
|
+
import type { InternalLogger } from '../../logger/internal-logger'
|
|
12
|
+
import type { DebugOption } from '../../logger/types'
|
|
13
|
+
import type { AudioAdapter } from './adapter'
|
|
14
|
+
import type { AudioGenerationResult, StreamChunk } from '../../types'
|
|
15
|
+
|
|
16
|
+
// ===========================
|
|
17
|
+
// Activity Kind
|
|
18
|
+
// ===========================
|
|
19
|
+
|
|
20
|
+
/** The adapter kind this activity handles */
|
|
21
|
+
export const kind = 'audio' as const
|
|
22
|
+
|
|
23
|
+
// ===========================
|
|
24
|
+
// Type Extraction Helpers
|
|
25
|
+
// ===========================
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Extract provider options from an AudioAdapter via ~types.
|
|
29
|
+
*/
|
|
30
|
+
export type AudioProviderOptions<TAdapter> =
|
|
31
|
+
TAdapter extends AudioAdapter<any, any>
|
|
32
|
+
? TAdapter['~types']['providerOptions']
|
|
33
|
+
: object
|
|
34
|
+
|
|
35
|
+
// ===========================
|
|
36
|
+
// Activity Options Type
|
|
37
|
+
// ===========================
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Options for the audio generation activity.
|
|
41
|
+
* The model is extracted from the adapter's model property.
|
|
42
|
+
*
|
|
43
|
+
* @template TAdapter - The audio adapter type
|
|
44
|
+
* @template TStream - Whether to stream the output
|
|
45
|
+
*/
|
|
46
|
+
export interface AudioActivityOptions<
|
|
47
|
+
TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,
|
|
48
|
+
TStream extends boolean = false,
|
|
49
|
+
> {
|
|
50
|
+
/** The audio adapter to use (must be created with a model) */
|
|
51
|
+
adapter: TAdapter & { kind: typeof kind }
|
|
52
|
+
/** Text description of the desired audio */
|
|
53
|
+
prompt: string
|
|
54
|
+
/** Desired duration in seconds */
|
|
55
|
+
duration?: number
|
|
56
|
+
/** Provider-specific options for audio generation */
|
|
57
|
+
modelOptions?: AudioProviderOptions<TAdapter>
|
|
58
|
+
/**
|
|
59
|
+
* Whether to stream the generation result.
|
|
60
|
+
* When true, returns an AsyncIterable<StreamChunk> for streaming transport.
|
|
61
|
+
* When false or not provided, returns a Promise<AudioGenerationResult>.
|
|
62
|
+
*
|
|
63
|
+
* @default false
|
|
64
|
+
*/
|
|
65
|
+
stream?: TStream
|
|
66
|
+
/**
|
|
67
|
+
* Enable debug logging. Pass `true` to enable all categories, `false` to
|
|
68
|
+
* silence everything including errors, or a `DebugConfig` object for granular
|
|
69
|
+
* control and/or a custom `Logger`.
|
|
70
|
+
*/
|
|
71
|
+
debug?: DebugOption
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// ===========================
|
|
75
|
+
// Activity Result Type
|
|
76
|
+
// ===========================
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Result type for the audio generation activity.
|
|
80
|
+
* - If stream is true: AsyncIterable<StreamChunk>
|
|
81
|
+
* - Otherwise: Promise<AudioGenerationResult>
|
|
82
|
+
*/
|
|
83
|
+
export type AudioActivityResult<TStream extends boolean = false> =
|
|
84
|
+
TStream extends true
|
|
85
|
+
? AsyncIterable<StreamChunk>
|
|
86
|
+
: Promise<AudioGenerationResult>
|
|
87
|
+
|
|
88
|
+
function createId(prefix: string): string {
|
|
89
|
+
return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// ===========================
|
|
93
|
+
// Activity Implementation
|
|
94
|
+
// ===========================
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Audio generation activity - generates audio from text prompts.
|
|
98
|
+
*
|
|
99
|
+
* Uses AI models to create music, sound effects, and other audio content.
|
|
100
|
+
*
|
|
101
|
+
* @example Generate music from a prompt
|
|
102
|
+
* ```ts
|
|
103
|
+
* import { generateAudio } from '@tanstack/ai'
|
|
104
|
+
* import { falAudio } from '@tanstack/ai-fal'
|
|
105
|
+
*
|
|
106
|
+
* const result = await generateAudio({
|
|
107
|
+
* adapter: falAudio('fal-ai/diffrhythm'),
|
|
108
|
+
* prompt: 'An upbeat electronic track with synths',
|
|
109
|
+
* duration: 10
|
|
110
|
+
* })
|
|
111
|
+
*
|
|
112
|
+
* console.log(result.audio.url) // URL to generated audio
|
|
113
|
+
* ```
|
|
114
|
+
*/
|
|
115
|
+
export function generateAudio<
|
|
116
|
+
TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,
|
|
117
|
+
TStream extends boolean = false,
|
|
118
|
+
>(
|
|
119
|
+
options: AudioActivityOptions<TAdapter, TStream>,
|
|
120
|
+
): AudioActivityResult<TStream> {
|
|
121
|
+
if (options.stream) {
|
|
122
|
+
return streamGenerationResult(() =>
|
|
123
|
+
runGenerateAudio(options),
|
|
124
|
+
) as AudioActivityResult<TStream>
|
|
125
|
+
}
|
|
126
|
+
return runGenerateAudio(options) as AudioActivityResult<TStream>
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Run the core audio generation logic (non-streaming).
|
|
131
|
+
*/
|
|
132
|
+
async function runGenerateAudio<
|
|
133
|
+
TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,
|
|
134
|
+
>(
|
|
135
|
+
options: AudioActivityOptions<TAdapter, boolean>,
|
|
136
|
+
): Promise<AudioGenerationResult> {
|
|
137
|
+
const { adapter, stream: _stream, debug: _debug, ...rest } = options
|
|
138
|
+
const model = adapter.model
|
|
139
|
+
const requestId = createId('audio')
|
|
140
|
+
const startTime = Date.now()
|
|
141
|
+
const logger: InternalLogger = resolveDebugOption(options.debug)
|
|
142
|
+
const providerName =
|
|
143
|
+
(adapter as { name?: string; provider?: string }).provider ??
|
|
144
|
+
(adapter as { name?: string }).name ??
|
|
145
|
+
'unknown'
|
|
146
|
+
|
|
147
|
+
aiEventClient.emit('audio:request:started', {
|
|
148
|
+
requestId,
|
|
149
|
+
provider: adapter.name,
|
|
150
|
+
model,
|
|
151
|
+
prompt: rest.prompt,
|
|
152
|
+
duration: rest.duration,
|
|
153
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
154
|
+
timestamp: startTime,
|
|
155
|
+
})
|
|
156
|
+
|
|
157
|
+
logger.request(`activity=generateAudio provider=${providerName}`, {
|
|
158
|
+
provider: providerName,
|
|
159
|
+
model,
|
|
160
|
+
})
|
|
161
|
+
|
|
162
|
+
try {
|
|
163
|
+
const result = await adapter.generateAudio({ ...rest, model, logger })
|
|
164
|
+
const elapsedMs = Date.now() - startTime
|
|
165
|
+
|
|
166
|
+
aiEventClient.emit('audio:request:completed', {
|
|
167
|
+
requestId,
|
|
168
|
+
provider: adapter.name,
|
|
169
|
+
model,
|
|
170
|
+
audio: result.audio,
|
|
171
|
+
duration: elapsedMs,
|
|
172
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
173
|
+
timestamp: Date.now(),
|
|
174
|
+
})
|
|
175
|
+
|
|
176
|
+
logger.output(`activity=generateAudio provider=${providerName}`, {
|
|
177
|
+
contentType: result.audio.contentType,
|
|
178
|
+
audioDuration: result.audio.duration,
|
|
179
|
+
})
|
|
180
|
+
|
|
181
|
+
return result
|
|
182
|
+
} catch (error) {
|
|
183
|
+
const elapsedMs = Date.now() - startTime
|
|
184
|
+
const err = error as Error
|
|
185
|
+
aiEventClient.emit('audio:request:error', {
|
|
186
|
+
requestId,
|
|
187
|
+
provider: adapter.name,
|
|
188
|
+
model,
|
|
189
|
+
error: { message: err.message, name: err.name },
|
|
190
|
+
duration: elapsedMs,
|
|
191
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
192
|
+
timestamp: Date.now(),
|
|
193
|
+
})
|
|
194
|
+
logger.errors('generateAudio activity failed', {
|
|
195
|
+
error,
|
|
196
|
+
source: 'generateAudio',
|
|
197
|
+
})
|
|
198
|
+
throw error
|
|
199
|
+
}
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
// ===========================
|
|
203
|
+
// Options Factory
|
|
204
|
+
// ===========================
|
|
205
|
+
|
|
206
|
+
/**
|
|
207
|
+
* Create typed options for the generateAudio() function without executing.
|
|
208
|
+
*/
|
|
209
|
+
export function createAudioOptions<
|
|
210
|
+
TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,
|
|
211
|
+
TStream extends boolean = false,
|
|
212
|
+
>(
|
|
213
|
+
options: AudioActivityOptions<TAdapter, TStream>,
|
|
214
|
+
): AudioActivityOptions<TAdapter, TStream> {
|
|
215
|
+
return options
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
// Re-export adapter types
|
|
219
|
+
export type {
|
|
220
|
+
AudioAdapter,
|
|
221
|
+
AudioAdapterConfig,
|
|
222
|
+
AnyAudioAdapter,
|
|
223
|
+
} from './adapter'
|
|
224
|
+
export { BaseAudioAdapter } from './adapter'
|
|
@@ -96,7 +96,7 @@ export abstract class BaseImageAdapter<
|
|
|
96
96
|
|
|
97
97
|
protected config: ImageAdapterConfig
|
|
98
98
|
|
|
99
|
-
constructor(config: ImageAdapterConfig = {}
|
|
99
|
+
constructor(model: TModel, config: ImageAdapterConfig = {}) {
|
|
100
100
|
this.config = config
|
|
101
101
|
this.model = model
|
|
102
102
|
}
|
|
@@ -44,7 +44,7 @@ export type TTSProviderOptions<TAdapter> =
|
|
|
44
44
|
* @template TStream - Whether to stream the output
|
|
45
45
|
*/
|
|
46
46
|
export interface TTSActivityOptions<
|
|
47
|
-
TAdapter extends TTSAdapter<string,
|
|
47
|
+
TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
|
|
48
48
|
TStream extends boolean = false,
|
|
49
49
|
> {
|
|
50
50
|
/** The TTS adapter to use (must be created with a model) */
|
|
@@ -126,7 +126,7 @@ function createId(prefix: string): string {
|
|
|
126
126
|
* ```
|
|
127
127
|
*/
|
|
128
128
|
export function generateSpeech<
|
|
129
|
-
TAdapter extends TTSAdapter<string,
|
|
129
|
+
TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
|
|
130
130
|
TStream extends boolean = false,
|
|
131
131
|
>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {
|
|
132
132
|
if (options.stream) {
|
|
@@ -140,9 +140,9 @@ export function generateSpeech<
|
|
|
140
140
|
/**
|
|
141
141
|
* Run the core TTS generation logic (non-streaming).
|
|
142
142
|
*/
|
|
143
|
-
async function runGenerateSpeech<
|
|
144
|
-
|
|
145
|
-
): Promise<TTSResult> {
|
|
143
|
+
async function runGenerateSpeech<
|
|
144
|
+
TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
|
|
145
|
+
>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {
|
|
146
146
|
const { adapter, stream: _stream, debug: _debug, ...rest } = options
|
|
147
147
|
const model = adapter.model
|
|
148
148
|
const requestId = createId('speech')
|
|
@@ -172,7 +172,6 @@ async function runGenerateSpeech<TAdapter extends TTSAdapter<string, object>>(
|
|
|
172
172
|
|
|
173
173
|
try {
|
|
174
174
|
const result = await adapter.generateSpeech({ ...rest, model, logger })
|
|
175
|
-
|
|
176
175
|
const duration = Date.now() - startTime
|
|
177
176
|
|
|
178
177
|
aiEventClient.emit('speech:request:completed', {
|
|
@@ -195,6 +194,17 @@ async function runGenerateSpeech<TAdapter extends TTSAdapter<string, object>>(
|
|
|
195
194
|
|
|
196
195
|
return result
|
|
197
196
|
} catch (error) {
|
|
197
|
+
const duration = Date.now() - startTime
|
|
198
|
+
const err = error as Error
|
|
199
|
+
aiEventClient.emit('speech:request:error', {
|
|
200
|
+
requestId,
|
|
201
|
+
provider: adapter.name,
|
|
202
|
+
model,
|
|
203
|
+
error: { message: err.message, name: err.name },
|
|
204
|
+
duration,
|
|
205
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
206
|
+
timestamp: Date.now(),
|
|
207
|
+
})
|
|
198
208
|
logger.errors('generateSpeech activity failed', {
|
|
199
209
|
error,
|
|
200
210
|
source: 'generateSpeech',
|
|
@@ -211,7 +221,7 @@ async function runGenerateSpeech<TAdapter extends TTSAdapter<string, object>>(
|
|
|
211
221
|
* Create typed options for the generateSpeech() function without executing.
|
|
212
222
|
*/
|
|
213
223
|
export function createSpeechOptions<
|
|
214
|
-
TAdapter extends TTSAdapter<string,
|
|
224
|
+
TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,
|
|
215
225
|
TStream extends boolean = false,
|
|
216
226
|
>(
|
|
217
227
|
options: TTSActivityOptions<TAdapter, TStream>,
|
|
@@ -74,7 +74,7 @@ export abstract class BaseTranscriptionAdapter<
|
|
|
74
74
|
|
|
75
75
|
protected config: TranscriptionAdapterConfig
|
|
76
76
|
|
|
77
|
-
constructor(config: TranscriptionAdapterConfig = {}
|
|
77
|
+
constructor(model: TModel, config: TranscriptionAdapterConfig = {}) {
|
|
78
78
|
this.config = config
|
|
79
79
|
this.model = model
|
|
80
80
|
}
|