@tanstack/ai 0.9.2 → 0.10.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/stream/processor.js +3 -0
- package/dist/esm/activities/chat/stream/processor.js.map +1 -1
- package/dist/esm/index.d.ts +1 -0
- package/dist/esm/index.js +3 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/tool-registry.d.ts +81 -0
- package/dist/esm/tool-registry.js +49 -0
- package/dist/esm/tool-registry.js.map +1 -0
- package/package.json +6 -4
- package/skills/ai-core/SKILL.md +59 -0
- package/skills/ai-core/adapter-configuration/SKILL.md +283 -0
- package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +97 -0
- package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +102 -0
- package/skills/ai-core/adapter-configuration/references/grok-adapter.md +77 -0
- package/skills/ai-core/adapter-configuration/references/groq-adapter.md +106 -0
- package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +82 -0
- package/skills/ai-core/adapter-configuration/references/openai-adapter.md +95 -0
- package/skills/ai-core/adapter-configuration/references/openrouter-adapter.md +99 -0
- package/skills/ai-core/ag-ui-protocol/SKILL.md +232 -0
- package/skills/ai-core/chat-experience/SKILL.md +506 -0
- package/skills/ai-core/custom-backend-integration/SKILL.md +463 -0
- package/skills/ai-core/media-generation/SKILL.md +471 -0
- package/skills/ai-core/middleware/SKILL.md +336 -0
- package/skills/ai-core/structured-outputs/SKILL.md +203 -0
- package/skills/ai-core/tool-calling/SKILL.md +411 -0
- package/src/activities/chat/stream/processor.ts +7 -0
- package/src/index.ts +7 -0
- package/src/tool-registry.ts +150 -0
|
@@ -0,0 +1,471 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: ai-core/media-generation
|
|
3
|
+
description: >
|
|
4
|
+
Image, video, speech (TTS), and transcription generation using
|
|
5
|
+
activity-specific adapters: generateImage() with openaiImage/geminiImage,
|
|
6
|
+
generateVideo() with async polling, generateSpeech() with openaiSpeech,
|
|
7
|
+
generateTranscription() with openaiTranscription. React hooks:
|
|
8
|
+
useGenerateImage, useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
9
|
+
TanStack Start server function integration with toServerSentEventsResponse.
|
|
10
|
+
type: sub-skill
|
|
11
|
+
library: tanstack-ai
|
|
12
|
+
library_version: '0.10.0'
|
|
13
|
+
sources:
|
|
14
|
+
- 'TanStack/ai:docs/media/generations.md'
|
|
15
|
+
- 'TanStack/ai:docs/media/generation-hooks.md'
|
|
16
|
+
- 'TanStack/ai:docs/media/image-generation.md'
|
|
17
|
+
- 'TanStack/ai:docs/media/video-generation.md'
|
|
18
|
+
- 'TanStack/ai:docs/media/text-to-speech.md'
|
|
19
|
+
- 'TanStack/ai:docs/media/transcription.md'
|
|
20
|
+
---
|
|
21
|
+
|
|
22
|
+
# Media Generation
|
|
23
|
+
|
|
24
|
+
> **Dependency note:** This skill builds on ai-core. Read it first for critical rules.
|
|
25
|
+
|
|
26
|
+
All media activities (image, speech, transcription, video) follow the same
|
|
27
|
+
server/client architecture: a `generate*()` function on the server, an SSE
|
|
28
|
+
transport via `toServerSentEventsResponse()`, and a framework hook on the
|
|
29
|
+
client.
|
|
30
|
+
|
|
31
|
+
## Setup -- Image Generation End-to-End
|
|
32
|
+
|
|
33
|
+
### Server (API route or TanStack Start server function)
|
|
34
|
+
|
|
35
|
+
```typescript
|
|
36
|
+
// routes/api/generate/image.ts
|
|
37
|
+
import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
|
|
38
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
39
|
+
|
|
40
|
+
export async function POST(req: Request) {
|
|
41
|
+
const { prompt, size, numberOfImages } = await req.json()
|
|
42
|
+
|
|
43
|
+
const stream = generateImage({
|
|
44
|
+
adapter: openaiImage('gpt-image-1'),
|
|
45
|
+
prompt,
|
|
46
|
+
size,
|
|
47
|
+
numberOfImages,
|
|
48
|
+
stream: true,
|
|
49
|
+
})
|
|
50
|
+
|
|
51
|
+
return toServerSentEventsResponse(stream)
|
|
52
|
+
}
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
### Client (React)
|
|
56
|
+
|
|
57
|
+
```tsx
|
|
58
|
+
import { useGenerateImage, fetchServerSentEvents } from '@tanstack/ai-react'
|
|
59
|
+
import { useState } from 'react'
|
|
60
|
+
|
|
61
|
+
function ImageGenerator() {
|
|
62
|
+
const [prompt, setPrompt] = useState('')
|
|
63
|
+
const { generate, result, isLoading, error, reset } = useGenerateImage({
|
|
64
|
+
connection: fetchServerSentEvents('/api/generate/image'),
|
|
65
|
+
})
|
|
66
|
+
|
|
67
|
+
return (
|
|
68
|
+
<div>
|
|
69
|
+
<input
|
|
70
|
+
value={prompt}
|
|
71
|
+
onChange={(e) => setPrompt(e.target.value)}
|
|
72
|
+
placeholder="Describe an image..."
|
|
73
|
+
/>
|
|
74
|
+
<button
|
|
75
|
+
onClick={() => generate({ prompt })}
|
|
76
|
+
disabled={isLoading || !prompt.trim()}
|
|
77
|
+
>
|
|
78
|
+
{isLoading ? 'Generating...' : 'Generate'}
|
|
79
|
+
</button>
|
|
80
|
+
|
|
81
|
+
{error && <p>Error: {error.message}</p>}
|
|
82
|
+
|
|
83
|
+
{result?.images.map((img, i) => (
|
|
84
|
+
<img
|
|
85
|
+
key={i}
|
|
86
|
+
src={img.url || `data:image/png;base64,${img.b64Json}`}
|
|
87
|
+
alt={img.revisedPrompt || 'Generated image'}
|
|
88
|
+
/>
|
|
89
|
+
))}
|
|
90
|
+
|
|
91
|
+
{result && <button onClick={reset}>Clear</button>}
|
|
92
|
+
</div>
|
|
93
|
+
)
|
|
94
|
+
}
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
### TanStack Start: Server Function Streaming (recommended)
|
|
98
|
+
|
|
99
|
+
When using TanStack Start, return `toServerSentEventsResponse()` from a
|
|
100
|
+
server function. The client fetcher receives a `Response` and the hook
|
|
101
|
+
parses it as SSE automatically:
|
|
102
|
+
|
|
103
|
+
```typescript
|
|
104
|
+
// lib/server-functions.ts
|
|
105
|
+
import { createServerFn } from '@tanstack/react-start'
|
|
106
|
+
import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
|
|
107
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
108
|
+
|
|
109
|
+
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
110
|
+
.inputValidator((data: { prompt: string; model?: string }) => data)
|
|
111
|
+
.handler(({ data }) => {
|
|
112
|
+
return toServerSentEventsResponse(
|
|
113
|
+
generateImage({
|
|
114
|
+
adapter: openaiImage(data.model ?? 'gpt-image-1'),
|
|
115
|
+
prompt: data.prompt,
|
|
116
|
+
stream: true,
|
|
117
|
+
}),
|
|
118
|
+
)
|
|
119
|
+
})
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
```tsx
|
|
123
|
+
import { useGenerateImage } from '@tanstack/ai-react'
|
|
124
|
+
import { generateImageStreamFn } from '../lib/server-functions'
|
|
125
|
+
|
|
126
|
+
function ImageGenerator() {
|
|
127
|
+
const { generate, result, isLoading } = useGenerateImage({
|
|
128
|
+
fetcher: (input) => generateImageStreamFn({ data: input }),
|
|
129
|
+
})
|
|
130
|
+
|
|
131
|
+
return (
|
|
132
|
+
<button
|
|
133
|
+
onClick={() => generate({ prompt: 'A sunset over mountains' })}
|
|
134
|
+
disabled={isLoading}
|
|
135
|
+
>
|
|
136
|
+
{isLoading ? 'Generating...' : 'Generate'}
|
|
137
|
+
</button>
|
|
138
|
+
)
|
|
139
|
+
}
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
144
|
+
## Core Patterns
|
|
145
|
+
|
|
146
|
+
### 1. Image Generation
|
|
147
|
+
|
|
148
|
+
Supported adapters: `openaiImage` (dall-e-2, dall-e-3, gpt-image-1,
|
|
149
|
+
gpt-image-1-mini) and `geminiImage` (gemini-3.1-flash-image-preview,
|
|
150
|
+
imagen-4.0-generate-001, etc.).
|
|
151
|
+
|
|
152
|
+
```typescript
|
|
153
|
+
import { generateImage } from '@tanstack/ai'
|
|
154
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
155
|
+
import { geminiImage } from '@tanstack/ai-gemini'
|
|
156
|
+
|
|
157
|
+
// OpenAI with quality/background options
|
|
158
|
+
const openaiResult = await generateImage({
|
|
159
|
+
adapter: openaiImage('gpt-image-1'),
|
|
160
|
+
prompt: 'A cat wearing a hat',
|
|
161
|
+
size: '1024x1024',
|
|
162
|
+
numberOfImages: 2,
|
|
163
|
+
modelOptions: {
|
|
164
|
+
quality: 'high',
|
|
165
|
+
background: 'transparent',
|
|
166
|
+
outputFormat: 'png',
|
|
167
|
+
},
|
|
168
|
+
})
|
|
169
|
+
|
|
170
|
+
// Gemini native model with aspect-ratio sizes
|
|
171
|
+
const geminiResult = await generateImage({
|
|
172
|
+
adapter: geminiImage('gemini-3.1-flash-image-preview'),
|
|
173
|
+
prompt: 'A futuristic cityscape at night',
|
|
174
|
+
size: '16:9_4K',
|
|
175
|
+
})
|
|
176
|
+
|
|
177
|
+
// Gemini Imagen model
|
|
178
|
+
const imagenResult = await generateImage({
|
|
179
|
+
adapter: geminiImage('imagen-4.0-generate-001'),
|
|
180
|
+
prompt: 'A landscape photo',
|
|
181
|
+
modelOptions: { aspectRatio: '16:9' },
|
|
182
|
+
})
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
Result shape: `ImageGenerationResult` with `images` array where each entry
|
|
186
|
+
has `b64Json?`, `url?`, and `revisedPrompt?`. OpenAI image URLs expire
|
|
187
|
+
after 1 hour -- download or display immediately.
|
|
188
|
+
|
|
189
|
+
### 2. Text-to-Speech
|
|
190
|
+
|
|
191
|
+
Adapter: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview).
|
|
192
|
+
|
|
193
|
+
```typescript
|
|
194
|
+
import { generateSpeech } from '@tanstack/ai'
|
|
195
|
+
import { openaiSpeech } from '@tanstack/ai-openai'
|
|
196
|
+
|
|
197
|
+
const result = await generateSpeech({
|
|
198
|
+
adapter: openaiSpeech('tts-1-hd'),
|
|
199
|
+
text: 'Hello, welcome to TanStack AI!',
|
|
200
|
+
voice: 'alloy', // alloy | echo | fable | onyx | nova | shimmer | ash | ballad | coral | sage | verse
|
|
201
|
+
format: 'mp3', // mp3 | opus | aac | flac | wav | pcm
|
|
202
|
+
speed: 1.0, // 0.25 to 4.0
|
|
203
|
+
})
|
|
204
|
+
|
|
205
|
+
// result.audio is base64-encoded audio
|
|
206
|
+
// result.format is the output format string
|
|
207
|
+
// result.contentType is the MIME type (e.g. "audio/mpeg")
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
Client hook:
|
|
211
|
+
|
|
212
|
+
```tsx
|
|
213
|
+
import { useGenerateSpeech, fetchServerSentEvents } from '@tanstack/ai-react'
|
|
214
|
+
|
|
215
|
+
const { generate, result, isLoading } = useGenerateSpeech({
|
|
216
|
+
connection: fetchServerSentEvents('/api/generate/speech'),
|
|
217
|
+
})
|
|
218
|
+
|
|
219
|
+
// Trigger: generate({ text: 'Hello!', voice: 'alloy' })
|
|
220
|
+
// Play: <audio src={`data:audio/${result.format};base64,${result.audio}`} controls />
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
### 3. Audio Transcription
|
|
224
|
+
|
|
225
|
+
Adapter: `openaiTranscription` (whisper-1, gpt-4o-transcribe,
|
|
226
|
+
gpt-4o-mini-transcribe).
|
|
227
|
+
|
|
228
|
+
```typescript
|
|
229
|
+
import { generateTranscription } from '@tanstack/ai'
|
|
230
|
+
import { openaiTranscription } from '@tanstack/ai-openai'
|
|
231
|
+
|
|
232
|
+
const result = await generateTranscription({
|
|
233
|
+
adapter: openaiTranscription('whisper-1'),
|
|
234
|
+
audio: audioFile, // File, Blob, base64 string, or data URL
|
|
235
|
+
language: 'en',
|
|
236
|
+
responseFormat: 'verbose_json',
|
|
237
|
+
modelOptions: {
|
|
238
|
+
include: ['segment', 'word'],
|
|
239
|
+
},
|
|
240
|
+
})
|
|
241
|
+
|
|
242
|
+
// result.text -- full transcribed text
|
|
243
|
+
// result.language -- detected/specified language
|
|
244
|
+
// result.duration -- audio duration in seconds
|
|
245
|
+
// result.segments -- timestamped segments with optional word-level timestamps
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
Client hook:
|
|
249
|
+
|
|
250
|
+
```tsx
|
|
251
|
+
import { useTranscription, fetchServerSentEvents } from '@tanstack/ai-react'
|
|
252
|
+
|
|
253
|
+
const { generate, result, isLoading } = useTranscription({
|
|
254
|
+
connection: fetchServerSentEvents('/api/transcribe'),
|
|
255
|
+
})
|
|
256
|
+
|
|
257
|
+
// Trigger: generate({ audio: dataUrl, language: 'en' })
|
|
258
|
+
```
|
|
259
|
+
|
|
260
|
+
### 4. Video Generation (Experimental -- async polling)
|
|
261
|
+
|
|
262
|
+
Video generation uses a jobs/polling architecture. The server creates a job,
|
|
263
|
+
polls for status, and streams updates to the client.
|
|
264
|
+
|
|
265
|
+
```typescript
|
|
266
|
+
import {
|
|
267
|
+
generateVideo,
|
|
268
|
+
getVideoJobStatus,
|
|
269
|
+
toServerSentEventsResponse,
|
|
270
|
+
} from '@tanstack/ai'
|
|
271
|
+
import { openaiVideo } from '@tanstack/ai-openai'
|
|
272
|
+
|
|
273
|
+
// Non-streaming: manual polling loop
|
|
274
|
+
const { jobId } = await generateVideo({
|
|
275
|
+
adapter: openaiVideo('sora-2'),
|
|
276
|
+
prompt: 'A golden retriever playing in sunflowers',
|
|
277
|
+
size: '1280x720',
|
|
278
|
+
duration: 8,
|
|
279
|
+
})
|
|
280
|
+
|
|
281
|
+
let status = await getVideoJobStatus({ adapter: openaiVideo('sora-2'), jobId })
|
|
282
|
+
while (status.status !== 'completed' && status.status !== 'failed') {
|
|
283
|
+
await new Promise((r) => setTimeout(r, 5000))
|
|
284
|
+
status = await getVideoJobStatus({ adapter: openaiVideo('sora-2'), jobId })
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
// Streaming: server handles polling, client gets real-time updates
|
|
288
|
+
const stream = generateVideo({
|
|
289
|
+
adapter: openaiVideo('sora-2'),
|
|
290
|
+
prompt: 'A flying car over a city',
|
|
291
|
+
stream: true,
|
|
292
|
+
pollingInterval: 3000,
|
|
293
|
+
maxDuration: 600_000,
|
|
294
|
+
})
|
|
295
|
+
return toServerSentEventsResponse(stream)
|
|
296
|
+
```
|
|
297
|
+
|
|
298
|
+
Client hook with job tracking:
|
|
299
|
+
|
|
300
|
+
```tsx
|
|
301
|
+
import { useGenerateVideo, fetchServerSentEvents } from '@tanstack/ai-react'
|
|
302
|
+
|
|
303
|
+
const { generate, result, jobId, videoStatus, isLoading } = useGenerateVideo({
|
|
304
|
+
connection: fetchServerSentEvents('/api/generate/video'),
|
|
305
|
+
onJobCreated: (id) => console.log('Job created:', id),
|
|
306
|
+
onStatusUpdate: (status) =>
|
|
307
|
+
console.log(`${status.status} (${status.progress}%)`),
|
|
308
|
+
})
|
|
309
|
+
|
|
310
|
+
// videoStatus: { jobId, status, progress?, url?, error? }
|
|
311
|
+
// result (on completion): { url }
|
|
312
|
+
```
|
|
313
|
+
|
|
314
|
+
---
|
|
315
|
+
|
|
316
|
+
## Common Hook API
|
|
317
|
+
|
|
318
|
+
All generation hooks return the same shape:
|
|
319
|
+
|
|
320
|
+
| Property | Type | Description |
|
|
321
|
+
| ----------- | -------------------------- | ------------------------------------------------ |
|
|
322
|
+
| `generate` | `(input) => Promise<void>` | Trigger generation |
|
|
323
|
+
| `result` | `T \| null` | Result (optionally transformed via `onResult`) |
|
|
324
|
+
| `isLoading` | `boolean` | Whether generation is in progress |
|
|
325
|
+
| `error` | `Error \| undefined` | Current error |
|
|
326
|
+
| `status` | `GenerationClientState` | `'idle' \| 'generating' \| 'success' \| 'error'` |
|
|
327
|
+
| `stop` | `() => void` | Abort current generation |
|
|
328
|
+
| `reset` | `() => void` | Clear state, return to idle |
|
|
329
|
+
|
|
330
|
+
Provide either `connection` (streaming SSE transport) or `fetcher`
|
|
331
|
+
(direct async call / server function returning `Response`). Use `onResult`
|
|
332
|
+
to transform what is stored:
|
|
333
|
+
|
|
334
|
+
```tsx
|
|
335
|
+
const { result } = useGenerateSpeech({
|
|
336
|
+
connection: fetchServerSentEvents('/api/generate/speech'),
|
|
337
|
+
onResult: (raw) => ({
|
|
338
|
+
audioUrl: `data:${raw.contentType};base64,${raw.audio}`,
|
|
339
|
+
duration: raw.duration,
|
|
340
|
+
}),
|
|
341
|
+
})
|
|
342
|
+
// result is typed as { audioUrl: string; duration?: number } | null
|
|
343
|
+
```
|
|
344
|
+
|
|
345
|
+
---
|
|
346
|
+
|
|
347
|
+
## Common Mistakes
|
|
348
|
+
|
|
349
|
+
### a. HIGH: Using the removed `embedding()` function
|
|
350
|
+
|
|
351
|
+
The `embedding()` function and `openaiEmbed` adapter were removed in v0.5.0.
|
|
352
|
+
Agents trained on older code may still generate this pattern.
|
|
353
|
+
|
|
354
|
+
**Wrong:**
|
|
355
|
+
|
|
356
|
+
```typescript
|
|
357
|
+
import { embedding } from '@tanstack/ai'
|
|
358
|
+
import { openaiEmbed } from '@tanstack/ai-openai'
|
|
359
|
+
|
|
360
|
+
const result = await embedding({
|
|
361
|
+
adapter: openaiEmbed(),
|
|
362
|
+
model: 'text-embedding-3-small',
|
|
363
|
+
input: 'Hello, world!',
|
|
364
|
+
})
|
|
365
|
+
```
|
|
366
|
+
|
|
367
|
+
**Correct -- use the provider SDK directly:**
|
|
368
|
+
|
|
369
|
+
```typescript
|
|
370
|
+
import OpenAI from 'openai'
|
|
371
|
+
|
|
372
|
+
const openai = new OpenAI({ apiKey: process.env.OPENAI_API_KEY })
|
|
373
|
+
|
|
374
|
+
const result = await openai.embeddings.create({
|
|
375
|
+
model: 'text-embedding-3-small',
|
|
376
|
+
input: 'Hello, world!',
|
|
377
|
+
})
|
|
378
|
+
```
|
|
379
|
+
|
|
380
|
+
> Source: docs/migration/migration.md. Note: Fixed in v0.5.0 but agents
|
|
381
|
+
> trained on older code may still generate this pattern.
|
|
382
|
+
|
|
383
|
+
### b. HIGH: Forgetting `toServerSentEventsResponse` with TanStack Start server functions
|
|
384
|
+
|
|
385
|
+
When using TanStack Start server functions with `stream: true`, you MUST
|
|
386
|
+
wrap the stream with `toServerSentEventsResponse()`. Returning the raw
|
|
387
|
+
stream from a server function will not work.
|
|
388
|
+
|
|
389
|
+
**Wrong:**
|
|
390
|
+
|
|
391
|
+
```typescript
|
|
392
|
+
export const generateImageStreamFn = createServerFn({ method: 'POST' }).handler(
|
|
393
|
+
({ data }) => {
|
|
394
|
+
// BUG: returning raw stream -- client cannot parse this
|
|
395
|
+
return generateImage({
|
|
396
|
+
adapter: openaiImage('gpt-image-1'),
|
|
397
|
+
prompt: data.prompt,
|
|
398
|
+
stream: true,
|
|
399
|
+
})
|
|
400
|
+
},
|
|
401
|
+
)
|
|
402
|
+
```
|
|
403
|
+
|
|
404
|
+
**Correct:**
|
|
405
|
+
|
|
406
|
+
```typescript
|
|
407
|
+
import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
|
|
408
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
409
|
+
|
|
410
|
+
export const generateImageStreamFn = createServerFn({ method: 'POST' }).handler(
|
|
411
|
+
({ data }) => {
|
|
412
|
+
return toServerSentEventsResponse(
|
|
413
|
+
generateImage({
|
|
414
|
+
adapter: openaiImage('gpt-image-1'),
|
|
415
|
+
prompt: data.prompt,
|
|
416
|
+
stream: true,
|
|
417
|
+
}),
|
|
418
|
+
)
|
|
419
|
+
},
|
|
420
|
+
)
|
|
421
|
+
```
|
|
422
|
+
|
|
423
|
+
> Source: maintainer interview.
|
|
424
|
+
|
|
425
|
+
### c. MEDIUM: Not downloading OpenAI image URLs before they expire
|
|
426
|
+
|
|
427
|
+
OpenAI image URLs expire after 1 hour. If you store the URL and display it
|
|
428
|
+
later, the image will silently break. Always download or display the image
|
|
429
|
+
immediately, or convert to base64 for persistence.
|
|
430
|
+
|
|
431
|
+
```typescript
|
|
432
|
+
const result = await generateImage({
|
|
433
|
+
adapter: openaiImage('dall-e-3'),
|
|
434
|
+
prompt: 'A mountain landscape',
|
|
435
|
+
})
|
|
436
|
+
|
|
437
|
+
// GOOD: download immediately
|
|
438
|
+
for (const img of result.images) {
|
|
439
|
+
if (img.url) {
|
|
440
|
+
const response = await fetch(img.url)
|
|
441
|
+
const blob = await response.blob()
|
|
442
|
+
// Save blob to storage...
|
|
443
|
+
}
|
|
444
|
+
}
|
|
445
|
+
|
|
446
|
+
// GOOD: use b64Json when available (no expiration)
|
|
447
|
+
// gpt-image-1 returns b64Json by default
|
|
448
|
+
```
|
|
449
|
+
|
|
450
|
+
> Source: docs/media/image-generation.md.
|
|
451
|
+
|
|
452
|
+
### d. MEDIUM: Using `stream: true` for activities that do not support streaming
|
|
453
|
+
|
|
454
|
+
Not all generation activities support streaming. Passing `stream: true` to
|
|
455
|
+
an activity that does not support it may hang or produce unexpected results.
|
|
456
|
+
Check the activity documentation before enabling streaming. All built-in
|
|
457
|
+
activities (`generateImage`, `generateSpeech`, `generateTranscription`,
|
|
458
|
+
`generateVideo`, `summarize`) support `stream: true`, but custom
|
|
459
|
+
`useGeneration` setups may not.
|
|
460
|
+
|
|
461
|
+
> Source: docs/media/generations.md.
|
|
462
|
+
|
|
463
|
+
---
|
|
464
|
+
|
|
465
|
+
## Cross-References
|
|
466
|
+
|
|
467
|
+
- See also: **ai-core/adapter-configuration/SKILL.md** -- Each media
|
|
468
|
+
activity requires a specific activity adapter (e.g., `openaiImage` for
|
|
469
|
+
images, `openaiSpeech` for speech, `openaiTranscription` for transcription,
|
|
470
|
+
`openaiVideo` for video). The adapter-configuration skill covers provider
|
|
471
|
+
setup, API keys, and model selection.
|