@tanstack/ai 0.53.0 → 0.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/README.md +14 -13
  2. package/dist/esm/activities/chat/index.js +22 -4
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/messages.d.ts +21 -1
  5. package/dist/esm/activities/chat/messages.js +50 -1
  6. package/dist/esm/activities/chat/messages.js.map +1 -1
  7. package/dist/esm/activities/chat/stream/processor.d.ts +17 -0
  8. package/dist/esm/activities/chat/stream/processor.js +27 -0
  9. package/dist/esm/activities/chat/stream/processor.js.map +1 -1
  10. package/dist/esm/activities/generateLiveVideo/adapter.d.ts +69 -0
  11. package/dist/esm/activities/generateLiveVideo/adapter.js +23 -0
  12. package/dist/esm/activities/generateLiveVideo/adapter.js.map +1 -0
  13. package/dist/esm/activities/generateLiveVideo/index.d.ts +99 -0
  14. package/dist/esm/activities/generateLiveVideo/index.js +162 -0
  15. package/dist/esm/activities/generateLiveVideo/index.js.map +1 -0
  16. package/dist/esm/activities/generateVideo/index.js +3 -1
  17. package/dist/esm/activities/generateVideo/index.js.map +1 -1
  18. package/dist/esm/activities/generateWorld/adapter.d.ts +69 -0
  19. package/dist/esm/activities/generateWorld/adapter.js +23 -0
  20. package/dist/esm/activities/generateWorld/adapter.js.map +1 -0
  21. package/dist/esm/activities/generateWorld/index.d.ts +99 -0
  22. package/dist/esm/activities/generateWorld/index.js +162 -0
  23. package/dist/esm/activities/generateWorld/index.js.map +1 -0
  24. package/dist/esm/activities/index.d.ts +8 -2
  25. package/dist/esm/activities/index.js +11 -7
  26. package/dist/esm/activities/middleware/types.d.ts +1 -1
  27. package/dist/esm/client.d.ts +4 -2
  28. package/dist/esm/client.js +3 -1
  29. package/dist/esm/client.js.map +1 -1
  30. package/dist/esm/index.d.ts +4 -2
  31. package/dist/esm/index.js +3 -1
  32. package/dist/esm/middlewares/otel.js +3 -1
  33. package/dist/esm/middlewares/otel.js.map +1 -1
  34. package/dist/esm/types.d.ts +112 -0
  35. package/package.json +3 -3
  36. package/skills/ai-core/adapter-configuration/SKILL.md +91 -43
  37. package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +39 -21
  38. package/skills/ai-core/adapter-configuration/references/byteplus-adapter.md +5 -0
  39. package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +14 -6
  40. package/skills/ai-core/adapter-configuration/references/grok-adapter.md +33 -25
  41. package/skills/ai-core/adapter-configuration/references/groq-adapter.md +7 -2
  42. package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +25 -12
  43. package/skills/ai-core/adapter-configuration/references/openai-adapter.md +19 -9
  44. package/skills/ai-core/adapter-configuration/references/openrouter-adapter.md +34 -21
  45. package/skills/ai-core/ag-ui-protocol/SKILL.md +16 -10
  46. package/skills/ai-core/chat-experience/SKILL.md +228 -108
  47. package/skills/ai-core/client-persistence/SKILL.md +21 -9
  48. package/skills/ai-core/custom-backend-integration/SKILL.md +86 -52
  49. package/skills/ai-core/debug-logging/SKILL.md +100 -18
  50. package/skills/ai-core/locks/SKILL.md +35 -7
  51. package/skills/ai-core/media-generation/SKILL.md +136 -61
  52. package/skills/ai-core/middleware/SKILL.md +174 -69
  53. package/skills/ai-core/structured-outputs/SKILL.md +98 -49
  54. package/skills/ai-core/tool-calling/SKILL.md +245 -158
  55. package/src/activities/chat/index.ts +29 -7
  56. package/src/activities/chat/messages.ts +60 -0
  57. package/src/activities/chat/stream/processor.ts +31 -0
  58. package/src/activities/generateLiveVideo/adapter.ts +99 -0
  59. package/src/activities/generateLiveVideo/index.ts +339 -0
  60. package/src/activities/generateVideo/index.ts +3 -4
  61. package/src/activities/generateWorld/adapter.ts +96 -0
  62. package/src/activities/generateWorld/index.ts +339 -0
  63. package/src/activities/index.ts +44 -0
  64. package/src/activities/middleware/types.ts +2 -0
  65. package/src/client.ts +8 -0
  66. package/src/index.ts +8 -0
  67. package/src/middlewares/otel.ts +2 -0
  68. package/src/types.ts +128 -0
@@ -5,7 +5,7 @@ description: >
5
5
  activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
6
6
  generateAudio() with geminiAudio/falAudio, generateVideo() with async
7
7
  polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
8
- per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
8
+ per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech/elevenlabsSpeech,
9
9
  generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
10
10
  useGenerateImage, useGenerateAudio,
11
11
  useGenerateSpeech, useTranscription, useGenerateVideo.
@@ -20,6 +20,7 @@ sources:
20
20
  - 'TanStack/ai:docs/media/audio-generation.md'
21
21
  - 'TanStack/ai:docs/media/video-generation.md'
22
22
  - 'TanStack/ai:docs/media/text-to-speech.md'
23
+ - 'TanStack/ai:docs/adapters/elevenlabs.md'
23
24
  - 'TanStack/ai:docs/media/transcription.md'
24
25
  - 'TanStack/ai:docs/advanced/debug-logging.md'
25
26
  ---
@@ -110,9 +111,10 @@ parses it as SSE automatically:
110
111
  import { createServerFn } from '@tanstack/react-start'
111
112
  import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
112
113
  import { openaiImage } from '@tanstack/ai-openai'
114
+ import type { OpenAIImageModel } from '@tanstack/ai-openai'
113
115
 
114
116
  export const generateImageStreamFn = createServerFn({ method: 'POST' })
115
- .inputValidator((data: { prompt: string; model?: string }) => data)
117
+ .inputValidator((data: { prompt: string; model?: OpenAIImageModel }) => data)
116
118
  .handler(({ data }) => {
117
119
  return toServerSentEventsResponse(
118
120
  generateImage({
@@ -183,7 +185,7 @@ const openaiResult = await generateImage({
183
185
  modelOptions: {
184
186
  quality: 'high',
185
187
  background: 'transparent',
186
- outputFormat: 'png',
188
+ output_format: 'png',
187
189
  },
188
190
  })
189
191
 
@@ -250,10 +252,10 @@ await generateImage({
250
252
  adapter: openaiImage('gpt-image-2'),
251
253
  prompt: [
252
254
  { type: 'text', content: 'Replace the masked region with a tree' },
253
- { type: 'image', source: { type: 'url', value: photoUrl } },
255
+ { type: 'image', source: { type: 'url', value: 'https://…/photo.png' } },
254
256
  {
255
257
  type: 'image',
256
- source: { type: 'url', value: maskUrl },
258
+ source: { type: 'url', value: 'https://…/mask.png' },
257
259
  metadata: { role: 'mask' },
258
260
  },
259
261
  ],
@@ -267,11 +269,11 @@ import { falVideo } from '@tanstack/ai-fal'
267
269
  await generateVideo({
268
270
  adapter: falVideo('fal-ai/kling-video/v3/pro/image-to-video'),
269
271
  prompt: [
270
- { type: 'image', source: { type: 'url', value: firstFrameUrl } },
272
+ { type: 'image', source: { type: 'url', value: 'https://…/first.png' } },
271
273
  { type: 'text', content: 'Slow cinematic push-in' },
272
274
  {
273
275
  type: 'image',
274
- source: { type: 'url', value: lastFrameUrl },
276
+ source: { type: 'url', value: 'https://…/last.png' },
275
277
  metadata: { role: 'end_frame' },
276
278
  },
277
279
  ],
@@ -302,14 +304,14 @@ with `allowUrlFetch: true` on the adapter config
302
304
 
303
305
  **Provider support matrix:**
304
306
 
305
- | Provider | `generateImage` image parts | `generateVideo` image parts |
306
- | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
307
- | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
308
- | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
309
- | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
310
- | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). Starting frame and reference inputs cannot be combined. A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
311
- | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
312
- | Anthropic | n/a (no image generation API). | n/a |
307
+ | Provider | `generateImage` image parts | `generateVideo` image parts |
308
+ | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
309
+ | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
310
+ | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
311
+ | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
312
+ | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). On 1.5 a starting frame can be combined with reference inputs (it pins the first frame). A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
313
+ | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
314
+ | Anthropic | n/a (no image generation API). | n/a |
313
315
 
314
316
  Video and audio prompt parts follow the same `metadata.role` convention
315
317
  for video-to-video and lipsync flows on fal. Grok accepts one source
@@ -351,8 +353,14 @@ const { generate, result, isLoading } = useGenerateAudio({
351
353
 
352
354
  ### 3. Text-to-Speech
353
355
 
354
- Adapters: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview) and
355
- `byteplusSpeech` (`seed-audio-1.0`).
356
+ Adapters include `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview),
357
+ `byteplusSpeech` (`seed-audio-1.0`), and `elevenlabsSpeech` (`eleven_v3`).
358
+
359
+ `elevenlabsSpeech` accepts `format: 'mp3' | 'pcm' | 'opus' | 'wav'`.
360
+ WAV output contains 44.1 kHz, 16-bit mono PCM with a RIFF header.
361
+ AAC and FLAC requests throw before the API call.
362
+ An explicit `modelOptions.outputFormat` overrides `format` and returns
363
+ the selected provider format without WAV wrapping.
356
364
 
357
365
  > **BytePlus Seed Speech is a separate product from ModelArk** — it reads
358
366
  > **`BYTEPLUS_VOICE_API_KEY`**, not `ARK_API_KEY`, and an Ark key there fails
@@ -363,7 +371,10 @@ Adapters: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview) and
363
371
  > drops `voice`. Voice ids ending `_uranus_bigtts` are TTS 2.0,
364
372
  > `_mars_bigtts` / `_moon_bigtts` are TTS 1.0, and `*_emo_v2_*` are the 1.0
365
373
  > voices that accept emotion tags. Formats: `wav`, `mp3`, `pcm`, `ogg_opus`;
366
- > `watermark` is also available on `modelOptions`.
374
+ > `modelOptions.watermark` takes an object here, not a boolean:
375
+ > `{ aigc_watermark }` for an audible marker and `{ aigc_metadata: { enable } }`
376
+ > for header provenance. `watermark: true` is shorthand for
377
+ > `{ aigc_watermark: true }`.
367
378
 
368
379
  ```typescript
369
380
  import { generateSpeech } from '@tanstack/ai'
@@ -404,35 +415,58 @@ gpt-4o-mini-transcribe, gpt-4o-transcribe-diarize) and `byteplusTranscription`
404
415
 
405
416
  > **Capturing audio in the browser:** Use `useAudioRecorder` from `@tanstack/ai-react` to record directly in the browser, then pass the recording as the `audio` input to `generate()`, or use `recording.part` as a prompt part in chat/generation calls. No transcoding or extra dependencies required — the recorder returns the native browser format (`audio/webm` or `audio/mp4`). For transcription, wrap it as a `data:` URL so the provider gets the real content type; passing raw `recording.base64` makes the adapter assume `audio/mpeg` and mislabel the webm/mp4 bytes.
406
417
  >
407
- > ```typescript
408
- > const { isRecording, start, stop } = useAudioRecorder()
409
- > const { generate } = useTranscription({
410
- > connection: fetchServerSentEvents('/api/transcribe'),
411
- > })
412
- > // ...
413
- > const recording = await stop()
414
- > const mimeType = recording.mimeType.split(';')[0] // strip ;codecs=...
415
- > await generate({ audio: `data:${mimeType};base64,${recording.base64}` })
418
+ > ```tsx
419
+ > import {
420
+ > useAudioRecorder,
421
+ > useTranscription,
422
+ > fetchServerSentEvents,
423
+ > } from '@tanstack/ai-react'
424
+ >
425
+ > function VoiceNote() {
426
+ > const { isRecording, start, stop } = useAudioRecorder()
427
+ > const { generate } = useTranscription({
428
+ > connection: fetchServerSentEvents('/api/transcribe'),
429
+ > })
430
+ >
431
+ > async function finish() {
432
+ > const recording = await stop()
433
+ > const mimeType = recording.mimeType.split(';')[0] // strip ;codecs=...
434
+ > await generate({ audio: `data:${mimeType};base64,${recording.base64}` })
435
+ > }
436
+ >
437
+ > return (
438
+ > <button onClick={isRecording ? finish : start}>
439
+ > {isRecording ? 'Stop & transcribe' : 'Record'}
440
+ > </button>
441
+ > )
442
+ > }
416
443
  > ```
417
444
 
418
445
  ```typescript
419
- import { generateTranscription } from '@tanstack/ai'
446
+ // routes/api/transcribe.ts
447
+ import { generateTranscription, toServerSentEventsResponse } from '@tanstack/ai'
420
448
  import { openaiTranscription } from '@tanstack/ai-openai'
421
449
 
422
- const result = await generateTranscription({
423
- adapter: openaiTranscription('whisper-1'),
424
- audio: audioFile, // File, Blob, base64 string, or data URL
425
- language: 'en',
426
- responseFormat: 'verbose_json',
427
- modelOptions: {
428
- timestamp_granularities: ['word', 'segment'],
429
- },
430
- })
450
+ export async function POST(request: Request) {
451
+ // The client hook below posts { data: { audio: dataUrl, language } }
452
+ const { audio, language } = (await request.json()).data
453
+
454
+ const stream = generateTranscription({
455
+ adapter: openaiTranscription('whisper-1'),
456
+ audio, // File, Blob, base64 string, or data URL
457
+ language,
458
+ responseFormat: 'verbose_json',
459
+ modelOptions: {
460
+ timestamp_granularities: ['word', 'segment'],
461
+ },
462
+ stream: true,
463
+ })
431
464
 
432
- // result.text -- full transcribed text
433
- // result.language -- detected/specified language
434
- // result.duration -- audio duration in seconds
435
- // result.segments -- timestamped segments (word-level timestamps are in result.words)
465
+ // On the client, result.text is the transcript, result.language the
466
+ // detected language, result.duration the seconds, result.segments the
467
+ // timestamped segments (word-level timestamps are in result.words).
468
+ return toServerSentEventsResponse(stream)
469
+ }
436
470
  ```
437
471
 
438
472
  For speaker diarization, use `openaiTranscription('gpt-4o-transcribe-diarize')`.
@@ -486,14 +520,17 @@ while (status.status !== 'completed' && status.status !== 'failed') {
486
520
  }
487
521
 
488
522
  // Streaming: server handles polling, client gets real-time updates
489
- const stream = generateVideo({
490
- adapter: openaiVideo('sora-2'),
491
- prompt: 'A flying car over a city',
492
- stream: true,
493
- pollingInterval: 3000,
494
- maxDuration: 600_000,
495
- })
496
- return toServerSentEventsResponse(stream)
523
+ export async function POST(request: Request) {
524
+ const { prompt } = await request.json()
525
+ const stream = generateVideo({
526
+ adapter: openaiVideo('sora-2'),
527
+ prompt,
528
+ stream: true,
529
+ pollingInterval: 3000,
530
+ maxDuration: 600_000,
531
+ })
532
+ return toServerSentEventsResponse(stream)
533
+ }
497
534
  ```
498
535
 
499
536
  Google Veo (`@tanstack/ai-gemini`) uses the same jobs/polling flow. Its
@@ -505,6 +542,7 @@ Image prompt parts route by `metadata.role`: first un-roled /
505
542
  `'reference'` / `'character'` → `referenceImages`:
506
543
 
507
544
  ```typescript
545
+ import { generateVideo } from '@tanstack/ai'
508
546
  import { geminiVideo } from '@tanstack/ai-gemini'
509
547
 
510
548
  const adapter = geminiVideo('veo-3.1-generate-preview')
@@ -538,6 +576,7 @@ media). For conversational editing, pass a prior generation's `jobId` as
538
576
  on 2026-09-30.
539
577
 
540
578
  ```typescript
579
+ import { generateVideo } from '@tanstack/ai'
541
580
  import { geminiVideo } from '@tanstack/ai-gemini'
542
581
 
543
582
  const omni = geminiVideo('gemini-omni-1.1-flash')
@@ -590,6 +629,7 @@ from OpenRouter's published metadata, with the same `availableDurations()` /
590
629
  `snapDuration()` helpers:
591
630
 
592
631
  ```typescript
632
+ import { generateVideo } from '@tanstack/ai'
593
633
  import { openRouterVideo } from '@tanstack/ai-openrouter'
594
634
 
595
635
  const adapter = openRouterVideo('bytedance/seedance-2.0')
@@ -643,6 +683,7 @@ const result = await generateImage({
643
683
 
644
684
  // usage.billed.quantity is the priced quantity. Multiply by the endpoint unit
645
685
  // price (GET https://api.fal.ai/v1/models/pricing?endpoint_id=…) for exact cost.
686
+ const unitPrice = 0.025 // USD per unit, from the pricing endpoint
646
687
  if (result.usage?.billed) {
647
688
  const cost = result.usage.billed.quantity * unitPrice
648
689
  }
@@ -769,6 +810,8 @@ Provide either `connection` (streaming SSE transport) or `fetcher`
769
810
  to transform what is stored:
770
811
 
771
812
  ```tsx
813
+ import { useGenerateSpeech, fetchServerSentEvents } from '@tanstack/ai-react'
814
+
772
815
  const { result } = useGenerateSpeech({
773
816
  connection: fetchServerSentEvents('/api/generate/speech'),
774
817
  onResult: (raw) => ({
@@ -790,7 +833,7 @@ Agents trained on older code may still generate this pattern.
790
833
 
791
834
  **Wrong:**
792
835
 
793
- ```typescript
836
+ ```typescript ignore
794
837
  import { embedding } from '@tanstack/ai'
795
838
  import { openaiEmbed } from '@tanstack/ai-openai'
796
839
 
@@ -825,27 +868,34 @@ stream from a server function will not work.
825
868
 
826
869
  **Wrong:**
827
870
 
828
- ```typescript
829
- export const generateImageStreamFn = createServerFn({ method: 'POST' }).handler(
830
- ({ data }) => {
871
+ ```typescript ignore
872
+ import { createServerFn } from '@tanstack/react-start'
873
+ import { generateImage } from '@tanstack/ai'
874
+ import { openaiImage } from '@tanstack/ai-openai'
875
+
876
+ export const generateImageStreamFn = createServerFn({ method: 'POST' })
877
+ .inputValidator((data: { prompt: string }) => data)
878
+ .handler(({ data }) => {
831
879
  // BUG: returning raw stream -- client cannot parse this
880
+ // (also a type error: an AsyncIterable is not a valid server-function return)
832
881
  return generateImage({
833
882
  adapter: openaiImage('gpt-image-1'),
834
883
  prompt: data.prompt,
835
884
  stream: true,
836
885
  })
837
- },
838
- )
886
+ })
839
887
  ```
840
888
 
841
889
  **Correct:**
842
890
 
843
891
  ```typescript
892
+ import { createServerFn } from '@tanstack/react-start'
844
893
  import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
845
894
  import { openaiImage } from '@tanstack/ai-openai'
846
895
 
847
- export const generateImageStreamFn = createServerFn({ method: 'POST' }).handler(
848
- ({ data }) => {
896
+ export const generateImageStreamFn = createServerFn({ method: 'POST' })
897
+ .inputValidator((data: { prompt: string }) => data)
898
+ .handler(({ data }) => {
849
899
  return toServerSentEventsResponse(
850
900
  generateImage({
851
901
  adapter: openaiImage('gpt-image-1'),
@@ -853,8 +903,7 @@ export const generateImageStreamFn = createServerFn({ method: 'POST' }).handler(
853
903
  stream: true,
854
904
  }),
855
905
  )
856
- },
857
- )
906
+ })
858
907
  ```
859
908
 
860
909
  > Source: maintainer interview.
@@ -866,6 +915,9 @@ later, the image will silently break. Always download or display the image
866
915
  immediately, or convert to base64 for persistence.
867
916
 
868
917
  ```typescript
918
+ import { generateImage } from '@tanstack/ai'
919
+ import { openaiImage } from '@tanstack/ai-openai'
920
+
869
921
  const result = await generateImage({
870
922
  adapter: openaiImage('dall-e-3'),
871
923
  prompt: 'A mountain landscape',
@@ -904,7 +956,7 @@ Gemini's `GenerateContentConfig` (used by Lyria 3 Pro / Lyria 3 Clip) does
904
956
  returns 30-second `audio/mp3`; Lyria 3 Pro returns `audio/mp3`. These fields
905
957
  are not in `GeminiAudioProviderOptions` — don't reach for them via `as any`.
906
958
 
907
- ```typescript
959
+ ```typescript ignore
908
960
  // WRONG — both fields are silently ignored or rejected by the SDK
909
961
  generateAudio({
910
962
  adapter: geminiAudio('lyria-3-pro-preview'),
@@ -914,6 +966,11 @@ generateAudio({
914
966
  negativePrompt: 'vocals', // unsupported
915
967
  } as any,
916
968
  })
969
+ ```
970
+
971
+ ```typescript
972
+ import { generateAudio } from '@tanstack/ai'
973
+ import { geminiAudio } from '@tanstack/ai-gemini'
917
974
 
918
975
  // CORRECT — shape the prompt itself for what you want
919
976
  generateAudio({
@@ -934,6 +991,10 @@ model's native field like `music_length_ms` or `seconds_total`), but not
934
991
  for Lyria.
935
992
 
936
993
  ```typescript
994
+ import { generateAudio } from '@tanstack/ai'
995
+ import { geminiAudio } from '@tanstack/ai-gemini'
996
+ import { falAudio } from '@tanstack/ai-fal'
997
+
937
998
  // For Lyria: put length guidance in the prompt
938
999
  generateAudio({
939
1000
  adapter: geminiAudio('lyria-3-pro-preview'),
@@ -958,6 +1019,9 @@ generateAudio({
958
1019
  `as any`.
959
1020
 
960
1021
  ```typescript
1022
+ import { generateSpeech } from '@tanstack/ai'
1023
+ import { geminiSpeech } from '@tanstack/ai-gemini'
1024
+
961
1025
  generateSpeech({
962
1026
  adapter: geminiSpeech('gemini-2.5-pro-preview-tts'),
963
1027
  text: '[Alice] Hi. [Bob] Hello!',
@@ -988,7 +1052,7 @@ narrowed per model, so passing an image part to a text-only model
988
1052
  also throw a clear runtime error as a backstop, so users learn at call
989
1053
  time rather than getting silently wrong output.
990
1054
 
991
- ```typescript
1055
+ ```typescript ignore
992
1056
  // WRONG — dall-e-3 has no edit/inputs API; image parts are a type error
993
1057
  generateImage({
994
1058
  adapter: openaiImage('dall-e-3'),
@@ -1006,6 +1070,14 @@ generateImage({
1006
1070
  { type: 'image', source: { type: 'url', value: url } }, // ❌ type error
1007
1071
  ],
1008
1072
  })
1073
+ ```
1074
+
1075
+ ```typescript
1076
+ import { generateImage } from '@tanstack/ai'
1077
+ import { openaiImage } from '@tanstack/ai-openai'
1078
+ import { geminiImage } from '@tanstack/ai-gemini'
1079
+
1080
+ const url = 'https://…/photo.png'
1009
1081
 
1010
1082
  // CORRECT — use a model that supports image-conditioned generation
1011
1083
  generateImage({
@@ -1035,6 +1107,9 @@ same `debug?: DebugOption` option that `chat()` does. Reach for `debug`
1035
1107
  instead of wiring up logging middleware.
1036
1108
 
1037
1109
  ```typescript
1110
+ import { generateSpeech } from '@tanstack/ai'
1111
+ import { openaiSpeech } from '@tanstack/ai-openai'
1112
+
1038
1113
  // When a speech generation sounds wrong or a transcription returns garbage
1039
1114
  generateSpeech({
1040
1115
  adapter: openaiSpeech('tts-1'),