@tanstack/ai 0.53.0 → 0.55.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -13
- package/dist/esm/activities/chat/index.js +22 -4
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/messages.d.ts +21 -1
- package/dist/esm/activities/chat/messages.js +50 -1
- package/dist/esm/activities/chat/messages.js.map +1 -1
- package/dist/esm/activities/chat/stream/processor.d.ts +17 -0
- package/dist/esm/activities/chat/stream/processor.js +27 -0
- package/dist/esm/activities/chat/stream/processor.js.map +1 -1
- package/dist/esm/activities/generateLiveVideo/adapter.d.ts +69 -0
- package/dist/esm/activities/generateLiveVideo/adapter.js +23 -0
- package/dist/esm/activities/generateLiveVideo/adapter.js.map +1 -0
- package/dist/esm/activities/generateLiveVideo/index.d.ts +99 -0
- package/dist/esm/activities/generateLiveVideo/index.js +162 -0
- package/dist/esm/activities/generateLiveVideo/index.js.map +1 -0
- package/dist/esm/activities/generateVideo/index.js +3 -1
- package/dist/esm/activities/generateVideo/index.js.map +1 -1
- package/dist/esm/activities/generateWorld/adapter.d.ts +69 -0
- package/dist/esm/activities/generateWorld/adapter.js +23 -0
- package/dist/esm/activities/generateWorld/adapter.js.map +1 -0
- package/dist/esm/activities/generateWorld/index.d.ts +99 -0
- package/dist/esm/activities/generateWorld/index.js +162 -0
- package/dist/esm/activities/generateWorld/index.js.map +1 -0
- package/dist/esm/activities/index.d.ts +8 -2
- package/dist/esm/activities/index.js +11 -7
- package/dist/esm/activities/middleware/types.d.ts +1 -1
- package/dist/esm/client.d.ts +4 -2
- package/dist/esm/client.js +3 -1
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +4 -2
- package/dist/esm/index.js +3 -1
- package/dist/esm/middlewares/otel.js +3 -1
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/types.d.ts +112 -0
- package/package.json +3 -3
- package/skills/ai-core/adapter-configuration/SKILL.md +91 -43
- package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +39 -21
- package/skills/ai-core/adapter-configuration/references/byteplus-adapter.md +5 -0
- package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +14 -6
- package/skills/ai-core/adapter-configuration/references/grok-adapter.md +33 -25
- package/skills/ai-core/adapter-configuration/references/groq-adapter.md +7 -2
- package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +25 -12
- package/skills/ai-core/adapter-configuration/references/openai-adapter.md +19 -9
- package/skills/ai-core/adapter-configuration/references/openrouter-adapter.md +34 -21
- package/skills/ai-core/ag-ui-protocol/SKILL.md +16 -10
- package/skills/ai-core/chat-experience/SKILL.md +228 -108
- package/skills/ai-core/client-persistence/SKILL.md +21 -9
- package/skills/ai-core/custom-backend-integration/SKILL.md +86 -52
- package/skills/ai-core/debug-logging/SKILL.md +100 -18
- package/skills/ai-core/locks/SKILL.md +35 -7
- package/skills/ai-core/media-generation/SKILL.md +136 -61
- package/skills/ai-core/middleware/SKILL.md +174 -69
- package/skills/ai-core/structured-outputs/SKILL.md +98 -49
- package/skills/ai-core/tool-calling/SKILL.md +245 -158
- package/src/activities/chat/index.ts +29 -7
- package/src/activities/chat/messages.ts +60 -0
- package/src/activities/chat/stream/processor.ts +31 -0
- package/src/activities/generateLiveVideo/adapter.ts +99 -0
- package/src/activities/generateLiveVideo/index.ts +339 -0
- package/src/activities/generateVideo/index.ts +3 -4
- package/src/activities/generateWorld/adapter.ts +96 -0
- package/src/activities/generateWorld/index.ts +339 -0
- package/src/activities/index.ts +44 -0
- package/src/activities/middleware/types.ts +2 -0
- package/src/client.ts +8 -0
- package/src/index.ts +8 -0
- package/src/middlewares/otel.ts +2 -0
- package/src/types.ts +128 -0
|
@@ -5,7 +5,7 @@ description: >
|
|
|
5
5
|
activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
|
|
6
6
|
generateAudio() with geminiAudio/falAudio, generateVideo() with async
|
|
7
7
|
polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
|
|
8
|
-
per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
|
|
8
|
+
per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech/elevenlabsSpeech,
|
|
9
9
|
generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
|
|
10
10
|
useGenerateImage, useGenerateAudio,
|
|
11
11
|
useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
@@ -20,6 +20,7 @@ sources:
|
|
|
20
20
|
- 'TanStack/ai:docs/media/audio-generation.md'
|
|
21
21
|
- 'TanStack/ai:docs/media/video-generation.md'
|
|
22
22
|
- 'TanStack/ai:docs/media/text-to-speech.md'
|
|
23
|
+
- 'TanStack/ai:docs/adapters/elevenlabs.md'
|
|
23
24
|
- 'TanStack/ai:docs/media/transcription.md'
|
|
24
25
|
- 'TanStack/ai:docs/advanced/debug-logging.md'
|
|
25
26
|
---
|
|
@@ -110,9 +111,10 @@ parses it as SSE automatically:
|
|
|
110
111
|
import { createServerFn } from '@tanstack/react-start'
|
|
111
112
|
import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
|
|
112
113
|
import { openaiImage } from '@tanstack/ai-openai'
|
|
114
|
+
import type { OpenAIImageModel } from '@tanstack/ai-openai'
|
|
113
115
|
|
|
114
116
|
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
115
|
-
.inputValidator((data: { prompt: string; model?:
|
|
117
|
+
.inputValidator((data: { prompt: string; model?: OpenAIImageModel }) => data)
|
|
116
118
|
.handler(({ data }) => {
|
|
117
119
|
return toServerSentEventsResponse(
|
|
118
120
|
generateImage({
|
|
@@ -183,7 +185,7 @@ const openaiResult = await generateImage({
|
|
|
183
185
|
modelOptions: {
|
|
184
186
|
quality: 'high',
|
|
185
187
|
background: 'transparent',
|
|
186
|
-
|
|
188
|
+
output_format: 'png',
|
|
187
189
|
},
|
|
188
190
|
})
|
|
189
191
|
|
|
@@ -250,10 +252,10 @@ await generateImage({
|
|
|
250
252
|
adapter: openaiImage('gpt-image-2'),
|
|
251
253
|
prompt: [
|
|
252
254
|
{ type: 'text', content: 'Replace the masked region with a tree' },
|
|
253
|
-
{ type: 'image', source: { type: 'url', value:
|
|
255
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/photo.png' } },
|
|
254
256
|
{
|
|
255
257
|
type: 'image',
|
|
256
|
-
source: { type: 'url', value:
|
|
258
|
+
source: { type: 'url', value: 'https://…/mask.png' },
|
|
257
259
|
metadata: { role: 'mask' },
|
|
258
260
|
},
|
|
259
261
|
],
|
|
@@ -267,11 +269,11 @@ import { falVideo } from '@tanstack/ai-fal'
|
|
|
267
269
|
await generateVideo({
|
|
268
270
|
adapter: falVideo('fal-ai/kling-video/v3/pro/image-to-video'),
|
|
269
271
|
prompt: [
|
|
270
|
-
{ type: 'image', source: { type: 'url', value:
|
|
272
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/first.png' } },
|
|
271
273
|
{ type: 'text', content: 'Slow cinematic push-in' },
|
|
272
274
|
{
|
|
273
275
|
type: 'image',
|
|
274
|
-
source: { type: 'url', value:
|
|
276
|
+
source: { type: 'url', value: 'https://…/last.png' },
|
|
275
277
|
metadata: { role: 'end_frame' },
|
|
276
278
|
},
|
|
277
279
|
],
|
|
@@ -302,14 +304,14 @@ with `allowUrlFetch: true` on the adapter config
|
|
|
302
304
|
|
|
303
305
|
**Provider support matrix:**
|
|
304
306
|
|
|
305
|
-
| Provider | `generateImage` image parts | `generateVideo` image parts
|
|
306
|
-
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
307
|
-
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1.
|
|
308
|
-
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing).
|
|
309
|
-
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`.
|
|
310
|
-
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5).
|
|
311
|
-
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata.
|
|
312
|
-
| Anthropic | n/a (no image generation API). | n/a
|
|
307
|
+
| Provider | `generateImage` image parts | `generateVideo` image parts |
|
|
308
|
+
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
309
|
+
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
|
|
310
|
+
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
|
|
311
|
+
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
|
|
312
|
+
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). On 1.5 a starting frame can be combined with reference inputs (it pins the first frame). A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
|
|
313
|
+
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
|
|
314
|
+
| Anthropic | n/a (no image generation API). | n/a |
|
|
313
315
|
|
|
314
316
|
Video and audio prompt parts follow the same `metadata.role` convention
|
|
315
317
|
for video-to-video and lipsync flows on fal. Grok accepts one source
|
|
@@ -351,8 +353,14 @@ const { generate, result, isLoading } = useGenerateAudio({
|
|
|
351
353
|
|
|
352
354
|
### 3. Text-to-Speech
|
|
353
355
|
|
|
354
|
-
Adapters
|
|
355
|
-
`byteplusSpeech` (`seed-audio-1.0`).
|
|
356
|
+
Adapters include `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview),
|
|
357
|
+
`byteplusSpeech` (`seed-audio-1.0`), and `elevenlabsSpeech` (`eleven_v3`).
|
|
358
|
+
|
|
359
|
+
`elevenlabsSpeech` accepts `format: 'mp3' | 'pcm' | 'opus' | 'wav'`.
|
|
360
|
+
WAV output contains 44.1 kHz, 16-bit mono PCM with a RIFF header.
|
|
361
|
+
AAC and FLAC requests throw before the API call.
|
|
362
|
+
An explicit `modelOptions.outputFormat` overrides `format` and returns
|
|
363
|
+
the selected provider format without WAV wrapping.
|
|
356
364
|
|
|
357
365
|
> **BytePlus Seed Speech is a separate product from ModelArk** — it reads
|
|
358
366
|
> **`BYTEPLUS_VOICE_API_KEY`**, not `ARK_API_KEY`, and an Ark key there fails
|
|
@@ -363,7 +371,10 @@ Adapters: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview) and
|
|
|
363
371
|
> drops `voice`. Voice ids ending `_uranus_bigtts` are TTS 2.0,
|
|
364
372
|
> `_mars_bigtts` / `_moon_bigtts` are TTS 1.0, and `*_emo_v2_*` are the 1.0
|
|
365
373
|
> voices that accept emotion tags. Formats: `wav`, `mp3`, `pcm`, `ogg_opus`;
|
|
366
|
-
> `watermark`
|
|
374
|
+
> `modelOptions.watermark` takes an object here, not a boolean:
|
|
375
|
+
> `{ aigc_watermark }` for an audible marker and `{ aigc_metadata: { enable } }`
|
|
376
|
+
> for header provenance. `watermark: true` is shorthand for
|
|
377
|
+
> `{ aigc_watermark: true }`.
|
|
367
378
|
|
|
368
379
|
```typescript
|
|
369
380
|
import { generateSpeech } from '@tanstack/ai'
|
|
@@ -404,35 +415,58 @@ gpt-4o-mini-transcribe, gpt-4o-transcribe-diarize) and `byteplusTranscription`
|
|
|
404
415
|
|
|
405
416
|
> **Capturing audio in the browser:** Use `useAudioRecorder` from `@tanstack/ai-react` to record directly in the browser, then pass the recording as the `audio` input to `generate()`, or use `recording.part` as a prompt part in chat/generation calls. No transcoding or extra dependencies required — the recorder returns the native browser format (`audio/webm` or `audio/mp4`). For transcription, wrap it as a `data:` URL so the provider gets the real content type; passing raw `recording.base64` makes the adapter assume `audio/mpeg` and mislabel the webm/mp4 bytes.
|
|
406
417
|
>
|
|
407
|
-
> ```
|
|
408
|
-
>
|
|
409
|
-
>
|
|
410
|
-
>
|
|
411
|
-
>
|
|
412
|
-
>
|
|
413
|
-
>
|
|
414
|
-
>
|
|
415
|
-
>
|
|
418
|
+
> ```tsx
|
|
419
|
+
> import {
|
|
420
|
+
> useAudioRecorder,
|
|
421
|
+
> useTranscription,
|
|
422
|
+
> fetchServerSentEvents,
|
|
423
|
+
> } from '@tanstack/ai-react'
|
|
424
|
+
>
|
|
425
|
+
> function VoiceNote() {
|
|
426
|
+
> const { isRecording, start, stop } = useAudioRecorder()
|
|
427
|
+
> const { generate } = useTranscription({
|
|
428
|
+
> connection: fetchServerSentEvents('/api/transcribe'),
|
|
429
|
+
> })
|
|
430
|
+
>
|
|
431
|
+
> async function finish() {
|
|
432
|
+
> const recording = await stop()
|
|
433
|
+
> const mimeType = recording.mimeType.split(';')[0] // strip ;codecs=...
|
|
434
|
+
> await generate({ audio: `data:${mimeType};base64,${recording.base64}` })
|
|
435
|
+
> }
|
|
436
|
+
>
|
|
437
|
+
> return (
|
|
438
|
+
> <button onClick={isRecording ? finish : start}>
|
|
439
|
+
> {isRecording ? 'Stop & transcribe' : 'Record'}
|
|
440
|
+
> </button>
|
|
441
|
+
> )
|
|
442
|
+
> }
|
|
416
443
|
> ```
|
|
417
444
|
|
|
418
445
|
```typescript
|
|
419
|
-
|
|
446
|
+
// routes/api/transcribe.ts
|
|
447
|
+
import { generateTranscription, toServerSentEventsResponse } from '@tanstack/ai'
|
|
420
448
|
import { openaiTranscription } from '@tanstack/ai-openai'
|
|
421
449
|
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
450
|
+
export async function POST(request: Request) {
|
|
451
|
+
// The client hook below posts { data: { audio: dataUrl, language } }
|
|
452
|
+
const { audio, language } = (await request.json()).data
|
|
453
|
+
|
|
454
|
+
const stream = generateTranscription({
|
|
455
|
+
adapter: openaiTranscription('whisper-1'),
|
|
456
|
+
audio, // File, Blob, base64 string, or data URL
|
|
457
|
+
language,
|
|
458
|
+
responseFormat: 'verbose_json',
|
|
459
|
+
modelOptions: {
|
|
460
|
+
timestamp_granularities: ['word', 'segment'],
|
|
461
|
+
},
|
|
462
|
+
stream: true,
|
|
463
|
+
})
|
|
431
464
|
|
|
432
|
-
// result.text
|
|
433
|
-
// result.
|
|
434
|
-
//
|
|
435
|
-
|
|
465
|
+
// On the client, result.text is the transcript, result.language the
|
|
466
|
+
// detected language, result.duration the seconds, result.segments the
|
|
467
|
+
// timestamped segments (word-level timestamps are in result.words).
|
|
468
|
+
return toServerSentEventsResponse(stream)
|
|
469
|
+
}
|
|
436
470
|
```
|
|
437
471
|
|
|
438
472
|
For speaker diarization, use `openaiTranscription('gpt-4o-transcribe-diarize')`.
|
|
@@ -486,14 +520,17 @@ while (status.status !== 'completed' && status.status !== 'failed') {
|
|
|
486
520
|
}
|
|
487
521
|
|
|
488
522
|
// Streaming: server handles polling, client gets real-time updates
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
523
|
+
export async function POST(request: Request) {
|
|
524
|
+
const { prompt } = await request.json()
|
|
525
|
+
const stream = generateVideo({
|
|
526
|
+
adapter: openaiVideo('sora-2'),
|
|
527
|
+
prompt,
|
|
528
|
+
stream: true,
|
|
529
|
+
pollingInterval: 3000,
|
|
530
|
+
maxDuration: 600_000,
|
|
531
|
+
})
|
|
532
|
+
return toServerSentEventsResponse(stream)
|
|
533
|
+
}
|
|
497
534
|
```
|
|
498
535
|
|
|
499
536
|
Google Veo (`@tanstack/ai-gemini`) uses the same jobs/polling flow. Its
|
|
@@ -505,6 +542,7 @@ Image prompt parts route by `metadata.role`: first un-roled /
|
|
|
505
542
|
`'reference'` / `'character'` → `referenceImages`:
|
|
506
543
|
|
|
507
544
|
```typescript
|
|
545
|
+
import { generateVideo } from '@tanstack/ai'
|
|
508
546
|
import { geminiVideo } from '@tanstack/ai-gemini'
|
|
509
547
|
|
|
510
548
|
const adapter = geminiVideo('veo-3.1-generate-preview')
|
|
@@ -538,6 +576,7 @@ media). For conversational editing, pass a prior generation's `jobId` as
|
|
|
538
576
|
on 2026-09-30.
|
|
539
577
|
|
|
540
578
|
```typescript
|
|
579
|
+
import { generateVideo } from '@tanstack/ai'
|
|
541
580
|
import { geminiVideo } from '@tanstack/ai-gemini'
|
|
542
581
|
|
|
543
582
|
const omni = geminiVideo('gemini-omni-1.1-flash')
|
|
@@ -590,6 +629,7 @@ from OpenRouter's published metadata, with the same `availableDurations()` /
|
|
|
590
629
|
`snapDuration()` helpers:
|
|
591
630
|
|
|
592
631
|
```typescript
|
|
632
|
+
import { generateVideo } from '@tanstack/ai'
|
|
593
633
|
import { openRouterVideo } from '@tanstack/ai-openrouter'
|
|
594
634
|
|
|
595
635
|
const adapter = openRouterVideo('bytedance/seedance-2.0')
|
|
@@ -643,6 +683,7 @@ const result = await generateImage({
|
|
|
643
683
|
|
|
644
684
|
// usage.billed.quantity is the priced quantity. Multiply by the endpoint unit
|
|
645
685
|
// price (GET https://api.fal.ai/v1/models/pricing?endpoint_id=…) for exact cost.
|
|
686
|
+
const unitPrice = 0.025 // USD per unit, from the pricing endpoint
|
|
646
687
|
if (result.usage?.billed) {
|
|
647
688
|
const cost = result.usage.billed.quantity * unitPrice
|
|
648
689
|
}
|
|
@@ -769,6 +810,8 @@ Provide either `connection` (streaming SSE transport) or `fetcher`
|
|
|
769
810
|
to transform what is stored:
|
|
770
811
|
|
|
771
812
|
```tsx
|
|
813
|
+
import { useGenerateSpeech, fetchServerSentEvents } from '@tanstack/ai-react'
|
|
814
|
+
|
|
772
815
|
const { result } = useGenerateSpeech({
|
|
773
816
|
connection: fetchServerSentEvents('/api/generate/speech'),
|
|
774
817
|
onResult: (raw) => ({
|
|
@@ -790,7 +833,7 @@ Agents trained on older code may still generate this pattern.
|
|
|
790
833
|
|
|
791
834
|
**Wrong:**
|
|
792
835
|
|
|
793
|
-
```typescript
|
|
836
|
+
```typescript ignore
|
|
794
837
|
import { embedding } from '@tanstack/ai'
|
|
795
838
|
import { openaiEmbed } from '@tanstack/ai-openai'
|
|
796
839
|
|
|
@@ -825,27 +868,34 @@ stream from a server function will not work.
|
|
|
825
868
|
|
|
826
869
|
**Wrong:**
|
|
827
870
|
|
|
828
|
-
```typescript
|
|
829
|
-
|
|
830
|
-
|
|
871
|
+
```typescript ignore
|
|
872
|
+
import { createServerFn } from '@tanstack/react-start'
|
|
873
|
+
import { generateImage } from '@tanstack/ai'
|
|
874
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
875
|
+
|
|
876
|
+
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
877
|
+
.inputValidator((data: { prompt: string }) => data)
|
|
878
|
+
.handler(({ data }) => {
|
|
831
879
|
// BUG: returning raw stream -- client cannot parse this
|
|
880
|
+
// (also a type error: an AsyncIterable is not a valid server-function return)
|
|
832
881
|
return generateImage({
|
|
833
882
|
adapter: openaiImage('gpt-image-1'),
|
|
834
883
|
prompt: data.prompt,
|
|
835
884
|
stream: true,
|
|
836
885
|
})
|
|
837
|
-
}
|
|
838
|
-
)
|
|
886
|
+
})
|
|
839
887
|
```
|
|
840
888
|
|
|
841
889
|
**Correct:**
|
|
842
890
|
|
|
843
891
|
```typescript
|
|
892
|
+
import { createServerFn } from '@tanstack/react-start'
|
|
844
893
|
import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
|
|
845
894
|
import { openaiImage } from '@tanstack/ai-openai'
|
|
846
895
|
|
|
847
|
-
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
848
|
-
({
|
|
896
|
+
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
897
|
+
.inputValidator((data: { prompt: string }) => data)
|
|
898
|
+
.handler(({ data }) => {
|
|
849
899
|
return toServerSentEventsResponse(
|
|
850
900
|
generateImage({
|
|
851
901
|
adapter: openaiImage('gpt-image-1'),
|
|
@@ -853,8 +903,7 @@ export const generateImageStreamFn = createServerFn({ method: 'POST' }).handler(
|
|
|
853
903
|
stream: true,
|
|
854
904
|
}),
|
|
855
905
|
)
|
|
856
|
-
}
|
|
857
|
-
)
|
|
906
|
+
})
|
|
858
907
|
```
|
|
859
908
|
|
|
860
909
|
> Source: maintainer interview.
|
|
@@ -866,6 +915,9 @@ later, the image will silently break. Always download or display the image
|
|
|
866
915
|
immediately, or convert to base64 for persistence.
|
|
867
916
|
|
|
868
917
|
```typescript
|
|
918
|
+
import { generateImage } from '@tanstack/ai'
|
|
919
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
920
|
+
|
|
869
921
|
const result = await generateImage({
|
|
870
922
|
adapter: openaiImage('dall-e-3'),
|
|
871
923
|
prompt: 'A mountain landscape',
|
|
@@ -904,7 +956,7 @@ Gemini's `GenerateContentConfig` (used by Lyria 3 Pro / Lyria 3 Clip) does
|
|
|
904
956
|
returns 30-second `audio/mp3`; Lyria 3 Pro returns `audio/mp3`. These fields
|
|
905
957
|
are not in `GeminiAudioProviderOptions` — don't reach for them via `as any`.
|
|
906
958
|
|
|
907
|
-
```typescript
|
|
959
|
+
```typescript ignore
|
|
908
960
|
// WRONG — both fields are silently ignored or rejected by the SDK
|
|
909
961
|
generateAudio({
|
|
910
962
|
adapter: geminiAudio('lyria-3-pro-preview'),
|
|
@@ -914,6 +966,11 @@ generateAudio({
|
|
|
914
966
|
negativePrompt: 'vocals', // unsupported
|
|
915
967
|
} as any,
|
|
916
968
|
})
|
|
969
|
+
```
|
|
970
|
+
|
|
971
|
+
```typescript
|
|
972
|
+
import { generateAudio } from '@tanstack/ai'
|
|
973
|
+
import { geminiAudio } from '@tanstack/ai-gemini'
|
|
917
974
|
|
|
918
975
|
// CORRECT — shape the prompt itself for what you want
|
|
919
976
|
generateAudio({
|
|
@@ -934,6 +991,10 @@ model's native field like `music_length_ms` or `seconds_total`), but not
|
|
|
934
991
|
for Lyria.
|
|
935
992
|
|
|
936
993
|
```typescript
|
|
994
|
+
import { generateAudio } from '@tanstack/ai'
|
|
995
|
+
import { geminiAudio } from '@tanstack/ai-gemini'
|
|
996
|
+
import { falAudio } from '@tanstack/ai-fal'
|
|
997
|
+
|
|
937
998
|
// For Lyria: put length guidance in the prompt
|
|
938
999
|
generateAudio({
|
|
939
1000
|
adapter: geminiAudio('lyria-3-pro-preview'),
|
|
@@ -958,6 +1019,9 @@ generateAudio({
|
|
|
958
1019
|
`as any`.
|
|
959
1020
|
|
|
960
1021
|
```typescript
|
|
1022
|
+
import { generateSpeech } from '@tanstack/ai'
|
|
1023
|
+
import { geminiSpeech } from '@tanstack/ai-gemini'
|
|
1024
|
+
|
|
961
1025
|
generateSpeech({
|
|
962
1026
|
adapter: geminiSpeech('gemini-2.5-pro-preview-tts'),
|
|
963
1027
|
text: '[Alice] Hi. [Bob] Hello!',
|
|
@@ -988,7 +1052,7 @@ narrowed per model, so passing an image part to a text-only model
|
|
|
988
1052
|
also throw a clear runtime error as a backstop, so users learn at call
|
|
989
1053
|
time rather than getting silently wrong output.
|
|
990
1054
|
|
|
991
|
-
```typescript
|
|
1055
|
+
```typescript ignore
|
|
992
1056
|
// WRONG — dall-e-3 has no edit/inputs API; image parts are a type error
|
|
993
1057
|
generateImage({
|
|
994
1058
|
adapter: openaiImage('dall-e-3'),
|
|
@@ -1006,6 +1070,14 @@ generateImage({
|
|
|
1006
1070
|
{ type: 'image', source: { type: 'url', value: url } }, // ❌ type error
|
|
1007
1071
|
],
|
|
1008
1072
|
})
|
|
1073
|
+
```
|
|
1074
|
+
|
|
1075
|
+
```typescript
|
|
1076
|
+
import { generateImage } from '@tanstack/ai'
|
|
1077
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
1078
|
+
import { geminiImage } from '@tanstack/ai-gemini'
|
|
1079
|
+
|
|
1080
|
+
const url = 'https://…/photo.png'
|
|
1009
1081
|
|
|
1010
1082
|
// CORRECT — use a model that supports image-conditioned generation
|
|
1011
1083
|
generateImage({
|
|
@@ -1035,6 +1107,9 @@ same `debug?: DebugOption` option that `chat()` does. Reach for `debug`
|
|
|
1035
1107
|
instead of wiring up logging middleware.
|
|
1036
1108
|
|
|
1037
1109
|
```typescript
|
|
1110
|
+
import { generateSpeech } from '@tanstack/ai'
|
|
1111
|
+
import { openaiSpeech } from '@tanstack/ai-openai'
|
|
1112
|
+
|
|
1038
1113
|
// When a speech generation sounds wrong or a transcription returns garbage
|
|
1039
1114
|
generateSpeech({
|
|
1040
1115
|
adapter: openaiSpeech('tts-1'),
|