@tanstack/ai 0.29.0 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/dist/esm/activities/chat/adapter.d.ts +10 -0
  2. package/dist/esm/activities/chat/adapter.js +1 -0
  3. package/dist/esm/activities/chat/adapter.js.map +1 -1
  4. package/dist/esm/activities/chat/index.d.ts +3 -2
  5. package/dist/esm/activities/chat/index.js +13 -1
  6. package/dist/esm/activities/chat/index.js.map +1 -1
  7. package/dist/esm/activities/chat/middleware/builder.d.ts +46 -0
  8. package/dist/esm/activities/chat/middleware/builder.js +17 -0
  9. package/dist/esm/activities/chat/middleware/builder.js.map +1 -0
  10. package/dist/esm/activities/chat/middleware/capabilities.d.ts +93 -0
  11. package/dist/esm/activities/chat/middleware/capabilities.js +45 -0
  12. package/dist/esm/activities/chat/middleware/capabilities.js.map +1 -0
  13. package/dist/esm/activities/chat/middleware/compose.d.ts +10 -0
  14. package/dist/esm/activities/chat/middleware/compose.js +47 -0
  15. package/dist/esm/activities/chat/middleware/compose.js.map +1 -1
  16. package/dist/esm/activities/chat/middleware/define.d.ts +20 -0
  17. package/dist/esm/activities/chat/middleware/define.js +7 -0
  18. package/dist/esm/activities/chat/middleware/define.js.map +1 -0
  19. package/dist/esm/activities/chat/middleware/index.d.ts +8 -0
  20. package/dist/esm/activities/chat/middleware/types.d.ts +51 -0
  21. package/dist/esm/activities/chat/middleware/validate.d.ts +19 -0
  22. package/dist/esm/activities/chat/middleware/validate.js +30 -0
  23. package/dist/esm/activities/chat/middleware/validate.js.map +1 -0
  24. package/dist/esm/activities/generateImage/adapter.d.ts +8 -4
  25. package/dist/esm/activities/generateImage/adapter.js.map +1 -1
  26. package/dist/esm/activities/generateImage/index.d.ts +19 -3
  27. package/dist/esm/activities/generateImage/index.js +12 -1
  28. package/dist/esm/activities/generateImage/index.js.map +1 -1
  29. package/dist/esm/activities/generateVideo/adapter.d.ts +65 -6
  30. package/dist/esm/activities/generateVideo/adapter.js +14 -0
  31. package/dist/esm/activities/generateVideo/adapter.js.map +1 -1
  32. package/dist/esm/activities/generateVideo/index.d.ts +31 -5
  33. package/dist/esm/activities/generateVideo/index.js.map +1 -1
  34. package/dist/esm/activities/generateVideo/snap.d.ts +14 -0
  35. package/dist/esm/activities/generateVideo/snap.js +54 -0
  36. package/dist/esm/activities/generateVideo/snap.js.map +1 -0
  37. package/dist/esm/activities/index.d.ts +3 -2
  38. package/dist/esm/activities/index.js +2 -0
  39. package/dist/esm/activities/index.js.map +1 -1
  40. package/dist/esm/client.d.ts +1 -1
  41. package/dist/esm/client.js.map +1 -1
  42. package/dist/esm/index.d.ts +4 -0
  43. package/dist/esm/index.js +8 -0
  44. package/dist/esm/index.js.map +1 -1
  45. package/dist/esm/middlewares/otel.js +40 -12
  46. package/dist/esm/middlewares/otel.js.map +1 -1
  47. package/dist/esm/types.d.ts +96 -7
  48. package/dist/esm/utilities/media-prompt.d.ts +35 -0
  49. package/dist/esm/utilities/media-prompt.js +43 -0
  50. package/dist/esm/utilities/media-prompt.js.map +1 -0
  51. package/package.json +10 -2
  52. package/skills/ai-core/media-generation/SKILL.md +173 -3
  53. package/src/activities/chat/adapter.ts +11 -0
  54. package/src/activities/chat/index.ts +20 -1
  55. package/src/activities/chat/middleware/builder.ts +109 -0
  56. package/src/activities/chat/middleware/capabilities.ts +162 -0
  57. package/src/activities/chat/middleware/compose.ts +51 -0
  58. package/src/activities/chat/middleware/define.ts +34 -0
  59. package/src/activities/chat/middleware/index.ts +20 -0
  60. package/src/activities/chat/middleware/types.ts +60 -0
  61. package/src/activities/chat/middleware/validate.ts +55 -0
  62. package/src/activities/generateImage/adapter.ts +16 -3
  63. package/src/activities/generateImage/index.ts +48 -4
  64. package/src/activities/generateVideo/adapter.ts +80 -4
  65. package/src/activities/generateVideo/index.ts +53 -4
  66. package/src/activities/generateVideo/snap.ts +100 -0
  67. package/src/activities/index.ts +4 -0
  68. package/src/client.ts +4 -0
  69. package/src/index.ts +18 -0
  70. package/src/middlewares/otel.ts +57 -12
  71. package/src/types.ts +119 -6
  72. package/src/utilities/media-prompt.ts +86 -0
@@ -1189,6 +1189,76 @@ export interface SummarizationResult {
1189
1189
  summary: string;
1190
1190
  usage: TokenUsage;
1191
1191
  }
1192
+ /**
1193
+ * Optional role hint on a media input part (image / video / audio). Adapters
1194
+ * read `metadata.role` to route the part to the provider-specific request
1195
+ * field — e.g. `'mask'` → OpenAI `mask` / fal `mask_url`, `'end_frame'` → fal
1196
+ * `end_image_url`, `'reference'` → fal `reference_image_urls`. When omitted
1197
+ * the adapter falls back to positional routing.
1198
+ */
1199
+ export type MediaInputRole = 'reference' | 'mask' | 'control' | 'start_frame' | 'end_frame' | 'character';
1200
+ /**
1201
+ * Metadata convention for image / video / audio inputs to media generation.
1202
+ * Carried on `ImagePart.metadata` / `VideoPart.metadata` / `AudioPart.metadata`
1203
+ * when used as conditioning inputs to `generateImage()` or `generateVideo()`.
1204
+ */
1205
+ export interface MediaInputMetadata {
1206
+ /** Optional role hint disambiguating the part's intent for the adapter */
1207
+ role?: MediaInputRole;
1208
+ /**
1209
+ * Optional user-defined label for this input (e.g. `'woman-in-red-dress'`).
1210
+ * **Informational only** — adapters never read it and the SDK never
1211
+ * rewrites prompt text based on it. Use it to correlate parts with the
1212
+ * references you write in your prompt using the provider's own syntax
1213
+ * (fal's `@Image1`, OpenAI's "image 1", etc.), or for your own
1214
+ * bookkeeping/logging.
1215
+ */
1216
+ tag?: string;
1217
+ }
1218
+ /**
1219
+ * A single part of a multimodal media-generation prompt. Reuses the chat
1220
+ * content-part shapes: text parts carry the instruction, image / video /
1221
+ * audio parts carry conditioning inputs (with an optional
1222
+ * `metadata.role` hint — see {@link MediaInputRole}).
1223
+ */
1224
+ export type MediaPromptPart = TextPart | ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata> | AudioPart<MediaInputMetadata>;
1225
+ /**
1226
+ * Prompt accepted by `generateImage()` / `generateVideo()`: a plain string,
1227
+ * or an ordered array of content parts for image-conditioned generation
1228
+ * ("not like this *(image)*, more like this *(image)*"). Part order is
1229
+ * meaningful — adapters with native multimodal prompts (Gemini, OpenRouter)
1230
+ * preserve the interleaving; named-field providers (fal, OpenAI, xAI)
1231
+ * extract the media parts and flatten the text. Text is always sent
1232
+ * verbatim: to reference inputs from the prompt, write the provider's own
1233
+ * syntax yourself (e.g. fal's `@Image1`, OpenAI's "image 1"). An array may
1234
+ * be media-only (e.g. upscalers or pure img2img endpoints that take no
1235
+ * instruction text).
1236
+ */
1237
+ export type MediaPrompt = string | Array<MediaPromptPart>;
1238
+ /**
1239
+ * Non-text modalities a media-generation model can accept in its prompt.
1240
+ */
1241
+ export type MediaPromptModality = 'image' | 'video' | 'audio';
1242
+ /** Maps a prompt modality to its content-part type. @internal */
1243
+ interface MediaPartByModality {
1244
+ image: ImagePart<MediaInputMetadata>;
1245
+ video: VideoPart<MediaInputMetadata>;
1246
+ audio: AudioPart<MediaInputMetadata>;
1247
+ }
1248
+ /**
1249
+ * Prompt type narrowed to the modalities a specific model supports.
1250
+ * `MediaPromptFor<never>` (a text-only model) is `string | Array<TextPart>`;
1251
+ * `MediaPromptFor<'image'>` additionally admits image parts, etc. Used by
1252
+ * the activity option types together with the adapter's per-model input
1253
+ * modality map so unsupported parts fail at compile time.
1254
+ */
1255
+ export type MediaPromptFor<TModalities extends MediaPromptModality = never> = string | Array<TextPart | MediaPartByModality[TModalities]>;
1256
+ /**
1257
+ * Per-model map from model name to the prompt modalities it accepts, used as
1258
+ * an adapter type parameter (`TModelInputModalitiesByName`). Models absent
1259
+ * from the map fall back to the unconstrained {@link MediaPrompt}.
1260
+ */
1261
+ export type ModelInputModalitiesByName = Record<string, ReadonlyArray<MediaPromptModality>>;
1192
1262
  /**
1193
1263
  * Options for image generation.
1194
1264
  * These are the common options supported across providers.
@@ -1196,8 +1266,16 @@ export interface SummarizationResult {
1196
1266
  export interface ImageGenerationOptions<TProviderOptions extends object = object, TSize extends string | undefined = string> {
1197
1267
  /** The model to use for image generation */
1198
1268
  model: string;
1199
- /** Text description of the desired image(s) */
1200
- prompt: string;
1269
+ /**
1270
+ * Description of the desired image(s): a plain string, or an ordered array
1271
+ * of content parts for image-conditioned generation (image-to-image,
1272
+ * reference-guided, edit, multi-reference). Media parts may carry
1273
+ * `metadata.role` to disambiguate intent (mask, control, reference, …).
1274
+ * Adapters map parts onto the provider-native request — e.g. Gemini
1275
+ * multimodal `contents`, OpenAI `images.edit()`, fal `image_url` /
1276
+ * `mask_url` — and throw a clear runtime error for unsupported modalities.
1277
+ */
1278
+ prompt: MediaPrompt;
1201
1279
  /** Number of images to generate (default: 1) */
1202
1280
  numberOfImages?: number;
1203
1281
  /** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
@@ -1293,15 +1371,26 @@ export interface AudioGenerationResult {
1293
1371
  *
1294
1372
  * @experimental Video generation is an experimental feature and may change.
1295
1373
  */
1296
- export interface VideoGenerationOptions<TProviderOptions extends object = object, TSize extends string | undefined = string> {
1374
+ export interface VideoGenerationOptions<TProviderOptions extends object = object, TSize extends string | undefined = string, TDuration extends string | number | undefined = number> {
1297
1375
  /** The model to use for video generation */
1298
1376
  model: string;
1299
- /** Text description of the desired video */
1300
- prompt: string;
1377
+ /**
1378
+ * Description of the desired video: a plain string, or an ordered array of
1379
+ * content parts for image-conditioned generation. Image parts may carry
1380
+ * `metadata.role` (`'start_frame' | 'end_frame' | 'reference' |
1381
+ * 'character'`) to disambiguate intent; adapters route them onto the
1382
+ * provider-native request (e.g. OpenAI Sora `input_reference`, fal
1383
+ * `image_url` / `end_image_url`) and throw at runtime if unsupported.
1384
+ */
1385
+ prompt: MediaPrompt;
1301
1386
  /** Video size — format depends on the provider (e.g., "16:9", "1280x720") */
1302
1387
  size?: TSize;
1303
- /** Video duration in seconds */
1304
- duration?: number;
1388
+ /**
1389
+ * Video duration in seconds. Adapters that declare a per-model duration
1390
+ * map narrow this to the model's valid union; use
1391
+ * `adapter.snapDuration(seconds)` to coerce raw seconds to a valid value.
1392
+ */
1393
+ duration?: TDuration;
1305
1394
  /** Model-specific options for video generation */
1306
1395
  modelOptions?: TProviderOptions;
1307
1396
  /**
@@ -0,0 +1,35 @@
1
+ import { AudioPart, ImagePart, MediaInputMetadata, MediaPrompt, MediaPromptPart, VideoPart } from '../types.js';
2
+ /**
3
+ * A {@link MediaPrompt} decomposed into the views adapters consume.
4
+ *
5
+ * Adapters with native multimodal prompts (Gemini `contents`, OpenRouter
6
+ * chat content parts) consume `parts` to preserve interleaving; named-field
7
+ * providers (fal, OpenAI) consume `text` plus the typed media buckets.
8
+ *
9
+ * Prompt text is **never rewritten**: text parts are concatenated verbatim.
10
+ * Providers that support referencing inputs from the prompt (e.g. fal's
11
+ * `@Image1`, OpenAI's "image 1" prose) expect the user to write that syntax
12
+ * themselves — the SDK does not inject or substitute markers.
13
+ */
14
+ export interface ResolvedMediaPrompt {
15
+ /**
16
+ * Text parts concatenated verbatim (paragraph-separated). Empty string
17
+ * for media-only prompts.
18
+ */
19
+ text: string;
20
+ /** The prompt as ordered parts; a string prompt becomes one text part. */
21
+ parts: Array<MediaPromptPart>;
22
+ /** Image parts in prompt order. */
23
+ images: Array<ImagePart<MediaInputMetadata>>;
24
+ /** Video parts in prompt order. */
25
+ videos: Array<VideoPart<MediaInputMetadata>>;
26
+ /** Audio parts in prompt order. */
27
+ audios: Array<AudioPart<MediaInputMetadata>>;
28
+ }
29
+ /**
30
+ * Decompose a {@link MediaPrompt} into flattened text and per-modality part
31
+ * buckets, preserving prompt order everywhere. This is the single downrev
32
+ * point from the canonical interleaved prompt shape to the named-field
33
+ * request shapes most providers expose.
34
+ */
35
+ export declare function resolveMediaPrompt(prompt: MediaPrompt): ResolvedMediaPrompt;
@@ -0,0 +1,43 @@
1
+ function resolveMediaPrompt(prompt) {
2
+ if (typeof prompt === "string") {
3
+ const textPart = { type: "text", content: prompt };
4
+ return {
5
+ text: prompt,
6
+ parts: [textPart],
7
+ images: [],
8
+ videos: [],
9
+ audios: []
10
+ };
11
+ }
12
+ const images = [];
13
+ const videos = [];
14
+ const audios = [];
15
+ const textSegments = [];
16
+ for (const part of prompt) {
17
+ switch (part.type) {
18
+ case "text":
19
+ if (part.content) textSegments.push(part.content);
20
+ break;
21
+ case "image":
22
+ images.push(part);
23
+ break;
24
+ case "video":
25
+ videos.push(part);
26
+ break;
27
+ case "audio":
28
+ audios.push(part);
29
+ break;
30
+ }
31
+ }
32
+ return {
33
+ text: textSegments.join("\n\n"),
34
+ parts: prompt,
35
+ images,
36
+ videos,
37
+ audios
38
+ };
39
+ }
40
+ export {
41
+ resolveMediaPrompt
42
+ };
43
+ //# sourceMappingURL=media-prompt.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"media-prompt.js","sources":["../../../src/utilities/media-prompt.ts"],"sourcesContent":["import type {\n AudioPart,\n ImagePart,\n MediaInputMetadata,\n MediaPrompt,\n MediaPromptPart,\n TextPart,\n VideoPart,\n} from '../types'\n\n/**\n * A {@link MediaPrompt} decomposed into the views adapters consume.\n *\n * Adapters with native multimodal prompts (Gemini `contents`, OpenRouter\n * chat content parts) consume `parts` to preserve interleaving; named-field\n * providers (fal, OpenAI) consume `text` plus the typed media buckets.\n *\n * Prompt text is **never rewritten**: text parts are concatenated verbatim.\n * Providers that support referencing inputs from the prompt (e.g. fal's\n * `@Image1`, OpenAI's \"image 1\" prose) expect the user to write that syntax\n * themselves — the SDK does not inject or substitute markers.\n */\nexport interface ResolvedMediaPrompt {\n /**\n * Text parts concatenated verbatim (paragraph-separated). Empty string\n * for media-only prompts.\n */\n text: string\n /** The prompt as ordered parts; a string prompt becomes one text part. */\n parts: Array<MediaPromptPart>\n /** Image parts in prompt order. */\n images: Array<ImagePart<MediaInputMetadata>>\n /** Video parts in prompt order. */\n videos: Array<VideoPart<MediaInputMetadata>>\n /** Audio parts in prompt order. */\n audios: Array<AudioPart<MediaInputMetadata>>\n}\n\n/**\n * Decompose a {@link MediaPrompt} into flattened text and per-modality part\n * buckets, preserving prompt order everywhere. This is the single downrev\n * point from the canonical interleaved prompt shape to the named-field\n * request shapes most providers expose.\n */\nexport function resolveMediaPrompt(prompt: MediaPrompt): ResolvedMediaPrompt {\n if (typeof prompt === 'string') {\n const textPart: TextPart = { type: 'text', content: prompt }\n return {\n text: prompt,\n parts: [textPart],\n images: [],\n videos: [],\n audios: [],\n }\n }\n\n const images: Array<ImagePart<MediaInputMetadata>> = []\n const videos: Array<VideoPart<MediaInputMetadata>> = []\n const audios: Array<AudioPart<MediaInputMetadata>> = []\n const textSegments: Array<string> = []\n\n for (const part of prompt) {\n switch (part.type) {\n case 'text':\n if (part.content) textSegments.push(part.content)\n break\n case 'image':\n images.push(part)\n break\n case 'video':\n videos.push(part)\n break\n case 'audio':\n audios.push(part)\n break\n }\n }\n\n return {\n text: textSegments.join('\\n\\n'),\n parts: prompt,\n images,\n videos,\n audios,\n }\n}\n"],"names":[],"mappings":"AA4CO,SAAS,mBAAmB,QAA0C;AAC3E,MAAI,OAAO,WAAW,UAAU;AAC9B,UAAM,WAAqB,EAAE,MAAM,QAAQ,SAAS,OAAA;AACpD,WAAO;AAAA,MACL,MAAM;AAAA,MACN,OAAO,CAAC,QAAQ;AAAA,MAChB,QAAQ,CAAA;AAAA,MACR,QAAQ,CAAA;AAAA,MACR,QAAQ,CAAA;AAAA,IAAC;AAAA,EAEb;AAEA,QAAM,SAA+C,CAAA;AACrD,QAAM,SAA+C,CAAA;AACrD,QAAM,SAA+C,CAAA;AACrD,QAAM,eAA8B,CAAA;AAEpC,aAAW,QAAQ,QAAQ;AACzB,YAAQ,KAAK,MAAA;AAAA,MACX,KAAK;AACH,YAAI,KAAK,QAAS,cAAa,KAAK,KAAK,OAAO;AAChD;AAAA,MACF,KAAK;AACH,eAAO,KAAK,IAAI;AAChB;AAAA,MACF,KAAK;AACH,eAAO,KAAK,IAAI;AAChB;AAAA,MACF,KAAK;AACH,eAAO,KAAK,IAAI;AAChB;AAAA,IAAA;AAAA,EAEN;AAEA,SAAO;AAAA,IACL,MAAM,aAAa,KAAK,MAAM;AAAA,IAC9B,OAAO;AAAA,IACP;AAAA,IACA;AAAA,IACA;AAAA,EAAA;AAEJ;"}
package/package.json CHANGED
@@ -1,14 +1,22 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.29.0",
3
+ "version": "0.32.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
7
+ "homepage": "https://tanstack.com/ai",
7
8
  "repository": {
8
9
  "type": "git",
9
10
  "url": "git+https://github.com/TanStack/ai.git",
10
11
  "directory": "packages/ai"
11
12
  },
13
+ "bugs": {
14
+ "url": "https://github.com/TanStack/ai/issues"
15
+ },
16
+ "funding": {
17
+ "type": "github",
18
+ "url": "https://github.com/sponsors/tannerlinsley"
19
+ },
12
20
  "type": "module",
13
21
  "module": "./dist/esm/index.js",
14
22
  "types": "./dist/esm/index.d.ts",
@@ -68,7 +76,7 @@
68
76
  "@ag-ui/core": "^0.0.52",
69
77
  "@standard-schema/spec": "^1.1.0",
70
78
  "partial-json": "^0.1.7",
71
- "@tanstack/ai-event-client": "0.6.0"
79
+ "@tanstack/ai-event-client": "0.6.3"
72
80
  },
73
81
  "peerDependencies": {
74
82
  "@opentelemetry/api": ">=1.9.0"
@@ -3,8 +3,9 @@ name: ai-core/media-generation
3
3
  description: >
4
4
  Image, audio, video, speech (TTS), and transcription generation using
5
5
  activity-specific adapters: generateImage() with openaiImage/geminiImage,
6
- generateAudio() with geminiAudio/falAudio, generateVideo() with async
7
- polling, generateSpeech() with openaiSpeech, generateTranscription() with
6
+ generateAudio() with geminiAudio/falAudio, generateVideo() with
7
+ openaiVideo/geminiVideo (async polling, per-model typed durations),
8
+ generateSpeech() with openaiSpeech, generateTranscription() with
8
9
  openaiTranscription. React hooks: useGenerateImage, useGenerateAudio,
9
10
  useGenerateSpeech, useTranscription, useGenerateVideo.
10
11
  TanStack Start server function integration with toServerSentEventsResponse.
@@ -189,6 +190,103 @@ Result shape: `ImageGenerationResult` with `images` array where each entry
189
190
  has `b64Json?`, `url?`, and `revisedPrompt?`. OpenAI image URLs expire
190
191
  after 1 hour -- download or display immediately.
191
192
 
193
+ #### Image-conditioned generation: multimodal `prompt` parts
194
+
195
+ Both `generateImage()` and `generateVideo()` accept the `prompt` either as
196
+ a plain string or as an ordered array of content parts (`TextPart` /
197
+ `ImagePart` / `VideoPart` / `AudioPart` — the same shapes used elsewhere in
198
+ TanStack AI). Part order is meaningful: natively multimodal providers
199
+ (Gemini, OpenRouter) receive parts in order; named-field providers (OpenAI,
200
+ fal, xAI) extract media parts and flatten the text. Prompt text is always
201
+ sent verbatim — to reference inputs from the prompt, write the provider's
202
+ own syntax (fal `@Image1`, OpenAI "image 1" prose); the SDK never injects
203
+ or rewrites markers. Each media part may carry an optional
204
+ `metadata.role` hint that adapters use to route the part to the
205
+ provider-specific field. The accepted part types are narrowed per model at
206
+ compile time via the adapter's input-modality map.
207
+
208
+ ```typescript
209
+ import { generateImage } from '@tanstack/ai'
210
+ import { openaiImage } from '@tanstack/ai-openai'
211
+
212
+ // Image-to-image (OpenAI gpt-image-2 / gpt-image-1, dall-e-2)
213
+ await generateImage({
214
+ adapter: openaiImage('gpt-image-2'),
215
+ prompt: [
216
+ { type: 'text', content: 'Turn this into a cinematic product photo' },
217
+ { type: 'image', source: { type: 'url', value: 'https://…/product.png' } },
218
+ ],
219
+ })
220
+
221
+ // Multi-reference (up to 16 for gpt-image models; up to ~14 for Gemini native
222
+ // — a provider limit, not enforced by the SDK)
223
+ await generateImage({
224
+ adapter: openaiImage('gpt-image-2'),
225
+ prompt: [
226
+ { type: 'text', content: 'Apply the second image as style to the first' },
227
+ { type: 'image', source: { type: 'url', value: 'https://…/product.png' } },
228
+ { type: 'image', source: { type: 'url', value: 'https://…/style.png' } },
229
+ ],
230
+ })
231
+
232
+ // Inpaint via metadata.role === 'mask' (OpenAI gpt-image models, dall-e-2; fal mask_url)
233
+ await generateImage({
234
+ adapter: openaiImage('gpt-image-2'),
235
+ prompt: [
236
+ { type: 'text', content: 'Replace the masked region with a tree' },
237
+ { type: 'image', source: { type: 'url', value: photoUrl } },
238
+ {
239
+ type: 'image',
240
+ source: { type: 'url', value: maskUrl },
241
+ metadata: { role: 'mask' },
242
+ },
243
+ ],
244
+ })
245
+
246
+ // Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional end_image_url)
247
+ import { generateVideo } from '@tanstack/ai'
248
+ import { falVideo } from '@tanstack/ai-fal'
249
+
250
+ await generateVideo({
251
+ adapter: falVideo('fal-ai/kling-video/v3/pro/image-to-video'),
252
+ prompt: [
253
+ { type: 'image', source: { type: 'url', value: firstFrameUrl } },
254
+ { type: 'text', content: 'Slow cinematic push-in' },
255
+ {
256
+ type: 'image',
257
+ source: { type: 'url', value: lastFrameUrl },
258
+ metadata: { role: 'end_frame' },
259
+ },
260
+ ],
261
+ })
262
+ ```
263
+
264
+ **Role hints** (`metadata.role`):
265
+
266
+ | Role | Maps to |
267
+ | --------------- | ----------------------------------------------------------------------------------------------------- |
268
+ | `'reference'` | fal `reference_image_urls`; Gemini multimodal part; positional otherwise |
269
+ | `'character'` | Same as `'reference'`; Veo `referenceImages` slot (planned — no Veo adapter yet) |
270
+ | `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
271
+ | `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
272
+ | `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); Veo `image` (planned) |
273
+ | `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); Veo `lastFrame` (planned) |
274
+
275
+ **Provider support matrix:**
276
+
277
+ | Provider | `generateImage` image parts | `generateVideo` image parts |
278
+ | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
279
+ | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
280
+ | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | No native Veo adapter yet — deferred to a follow-up. |
281
+ | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
282
+ | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
283
+ | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | n/a |
284
+ | Anthropic | n/a (no image generation API). | n/a |
285
+
286
+ Video and audio prompt parts follow the same `metadata.role` convention
287
+ for video-to-video and lipsync flows on fal; other providers throw when
288
+ they're passed.
289
+
192
290
  ### 2. Audio Generation (Music, Sound Effects)
193
291
 
194
292
  Distinct from TTS — `generateAudio()` produces non-speech audio content.
@@ -331,6 +429,31 @@ const stream = generateVideo({
331
429
  return toServerSentEventsResponse(stream)
332
430
  ```
333
431
 
432
+ Google Veo (`@tanstack/ai-gemini`) uses the same jobs/polling flow. Its
433
+ `duration` option is typed per model (e.g. `4 | 6 | 8` for Veo 3.x,
434
+ `5 | 6 | 8` for Veo 2); use `adapter.snapDuration(seconds)` to coerce raw
435
+ seconds and `adapter.availableDurations()` to enumerate the valid set.
436
+ Image prompt parts route by `metadata.role`: first un-roled /
437
+ `'start_frame'` image → input image, `'end_frame'` → `lastFrame`,
438
+ `'reference'` / `'character'` → `referenceImages`:
439
+
440
+ ```typescript
441
+ import { geminiVideo } from '@tanstack/ai-gemini'
442
+
443
+ const adapter = geminiVideo('veo-3.1-generate-preview')
444
+ adapter.availableDurations() // { kind: 'discrete', values: [4, 6, 8] }
445
+
446
+ const { jobId } = await generateVideo({
447
+ adapter,
448
+ prompt: 'A golden retriever playing in sunflowers',
449
+ size: '16:9', // Veo sizes are aspect ratios: '16:9' | '9:16'
450
+ duration: adapter.snapDuration(7), // 6
451
+ modelOptions: { resolution: '1080p', generateAudio: true },
452
+ })
453
+ // Note: Veo result URLs require the Google API key to download
454
+ // (x-goog-api-key header or ?key= query parameter).
455
+ ```
456
+
334
457
  Client hook with job tracking:
335
458
 
336
459
  ```tsx
@@ -607,7 +730,54 @@ generateSpeech({
607
730
 
608
731
  > Source: Gemini TTS adapter validation; CodeRabbit review of PR #463.
609
732
 
610
- ### h. LOW: Writing a logging middleware to see media chunks flow through
733
+ ### h. HIGH: Passing image prompt parts to a model that doesn't support image-conditioned generation
734
+
735
+ Not every model accepts image-conditioned prompts. The `prompt` type is
736
+ narrowed per model, so passing an image part to a text-only model
737
+ (dall-e-3, Imagen, grok-2-image) is a **compile-time error**; adapters
738
+ also throw a clear runtime error as a backstop, so users learn at call
739
+ time rather than getting silently wrong output.
740
+
741
+ ```typescript
742
+ // WRONG — dall-e-3 has no edit/inputs API; image parts are a type error
743
+ generateImage({
744
+ adapter: openaiImage('dall-e-3'),
745
+ prompt: [
746
+ { type: 'text', content: 'Edit this' },
747
+ { type: 'image', source: { type: 'url', value: url } }, // ❌ type error
748
+ ],
749
+ })
750
+
751
+ // WRONG — Imagen is text-to-image only; same compile-time rejection
752
+ generateImage({
753
+ adapter: geminiImage('imagen-4.0-generate-001'),
754
+ prompt: [
755
+ { type: 'text', content: 'Edit this' },
756
+ { type: 'image', source: { type: 'url', value: url } }, // ❌ type error
757
+ ],
758
+ })
759
+
760
+ // CORRECT — use a model that supports image-conditioned generation
761
+ generateImage({
762
+ adapter: openaiImage('gpt-image-2'), // edits up to 16 images
763
+ prompt: [
764
+ { type: 'text', content: 'Edit this' },
765
+ { type: 'image', source: { type: 'url', value: url } },
766
+ ],
767
+ })
768
+
769
+ generateImage({
770
+ adapter: geminiImage('gemini-3.1-flash-image-preview'), // native multimodal
771
+ prompt: [
772
+ { type: 'text', content: 'Edit this' },
773
+ { type: 'image', source: { type: 'url', value: url } },
774
+ ],
775
+ })
776
+ ```
777
+
778
+ > Source: docs/media/image-generation.md, docs/media/video-generation.md.
779
+
780
+ ### i. LOW: Writing a logging middleware to see media chunks flow through
611
781
 
612
782
  Every media activity — `generateAudio`, `generateSpeech`,
613
783
  `generateTranscription`, `generateImage`, `generateVideo` — accepts the
@@ -6,6 +6,7 @@ import type {
6
6
  TextOptions,
7
7
  TokenUsage,
8
8
  } from '../../types'
9
+ import type { CapabilityHandle } from './middleware/capabilities'
9
10
 
10
11
  /**
11
12
  * Configuration for adapter instances
@@ -79,6 +80,15 @@ export interface TextAdapter<
79
80
  /** The model this adapter is configured for */
80
81
  readonly model: TModel
81
82
 
83
+ /**
84
+ * Capabilities this adapter requires at runtime. `chat()` validates that the
85
+ * configured middleware provides each one. Model adapters omit this; harness
86
+ * adapters (e.g. a future `claudeCode()`) declare e.g. `[sandboxCapability]`.
87
+ * Runtime access to capabilities from inside the adapter is not yet wired —
88
+ * this is the declaration/validation surface only.
89
+ */
90
+ readonly requires?: ReadonlyArray<CapabilityHandle>
91
+
82
92
  /**
83
93
  * @internal Type-only properties for inference. Not assigned at runtime.
84
94
  */
@@ -183,6 +193,7 @@ export abstract class BaseTextAdapter<
183
193
  readonly kind = 'text' as const
184
194
  abstract readonly name: string
185
195
  readonly model: TModel
196
+ readonly requires?: ReadonlyArray<CapabilityHandle> = undefined
186
197
 
187
198
  // Type-only property - never assigned at runtime
188
199
  declare '~types': {
@@ -25,6 +25,8 @@ import {
25
25
  import { maxIterations as maxIterationsStrategy } from './agent-loop-strategies'
26
26
  import { convertMessagesToModelMessages, generateMessageId } from './messages'
27
27
  import { MiddlewareRunner } from './middleware/compose'
28
+ import { CapabilityRegistry } from './middleware/capabilities'
29
+ import { validateCapabilities } from './middleware/validate'
28
30
  import { MCPManager } from './mcp/manager'
29
31
  import type {
30
32
  ApprovalRequest,
@@ -54,11 +56,13 @@ import type {
54
56
  UIMessage,
55
57
  } from '../../types'
56
58
  import type {
59
+ AnyChatMiddleware,
57
60
  ChatMiddleware,
58
61
  ChatMiddlewareConfig,
59
62
  ChatMiddlewareContext,
60
63
  StructuredOutputMiddlewareConfig,
61
64
  } from './middleware/types'
65
+ import type { CheckCoverage } from './middleware/builder'
62
66
  import type { SystemPrompt } from '../../system-prompts'
63
67
  import type { InternalLogger } from '../../logger/internal-logger'
64
68
  import type { DebugOption } from '../../logger/types'
@@ -141,7 +145,8 @@ type TextActivityOptionsWithContext<
141
145
  'tools' | 'middleware' | 'context'
142
146
  > & {
143
147
  tools?: TTools
144
- middleware?: TMiddleware
148
+ middleware?: TMiddleware &
149
+ CheckCoverage<Extract<TMiddleware, ReadonlyArray<AnyChatMiddleware>>>
145
150
  } & RequiredContextFromInputs<TTools, TMiddleware>
146
151
 
147
152
  // ===========================
@@ -659,6 +664,15 @@ class TextEngine<
659
664
  // References
660
665
  messages: this.messages,
661
666
  createId: (prefix: string) => this.createId(prefix),
667
+ // Capability bookkeeping for this request (populated by middleware setup)
668
+ capabilities: new CapabilityRegistry(),
669
+ // Convenience accessors that delegate to a capability handle's own
670
+ // tuple getter/provider, keyed by this context. `getX(ctx)` and
671
+ // `ctx.get(X)` are interchangeable.
672
+ get: (capability) => capability[0](this.middlewareCtx),
673
+ getOptional: (capability) =>
674
+ capability[0](this.middlewareCtx, { optional: true }),
675
+ provide: (capability, value) => capability[1](this.middlewareCtx, value),
662
676
  }
663
677
  }
664
678
 
@@ -706,6 +720,9 @@ class TextEngine<
706
720
  })
707
721
 
708
722
  try {
723
+ // Provision capabilities before any consumer (onConfig onward) can read them
724
+ await this.middlewareRunner.runSetup(this.middlewareCtx)
725
+
709
726
  // Run initial onConfig (phase = init)
710
727
  this.middlewareCtx.phase = 'init'
711
728
  const initialConfig = this.buildMiddlewareConfig()
@@ -2564,6 +2581,8 @@ export function chat<
2564
2581
  TMiddleware
2565
2582
  >,
2566
2583
  ): TextActivityResult<TSchema, TStream> {
2584
+ validateCapabilities(options.middleware ?? [], options.adapter)
2585
+
2567
2586
  const { outputSchema, stream } = options
2568
2587
 
2569
2588
  // outputSchema + stream:true is the only branch that streams structured
@@ -0,0 +1,109 @@
1
+ import type { CapabilityHandle } from './capabilities'
2
+ import type { AnyChatMiddleware, ChatMiddleware } from './types'
3
+ import type { DefinedChatMiddleware } from './define'
4
+
5
+ /** Union of capability NAME literals from a tuple of handles. */
6
+ export type NamesOf<T extends ReadonlyArray<CapabilityHandle>> =
7
+ T[number]['capabilityName']
8
+
9
+ /** Names provided across a middleware array (imprecise middleware → `string`). */
10
+ export type ProvidedNames<TList extends ReadonlyArray<AnyChatMiddleware>> =
11
+ NonNullable<TList[number]['provides']> extends infer P
12
+ ? P extends ReadonlyArray<CapabilityHandle>
13
+ ? NamesOf<P>
14
+ : never
15
+ : never
16
+
17
+ /** Names required across a middleware array. */
18
+ export type RequiredNames<TList extends ReadonlyArray<AnyChatMiddleware>> =
19
+ NonNullable<TList[number]['requires']> extends infer P
20
+ ? P extends ReadonlyArray<CapabilityHandle>
21
+ ? NamesOf<P>
22
+ : never
23
+ : never
24
+
25
+ /**
26
+ * Branded marker surfaced when required capability names are missing from the
27
+ * provided set, so the compiler error names the gap instead of emitting an
28
+ * opaque "not assignable".
29
+ */
30
+ export type MissingCapabilities<TMissing extends string> = {
31
+ // The human-readable message lives in the property KEY, so TypeScript's
32
+ // "Property '<key>' is missing in type ... but required in type ..." error
33
+ // prints the explanation instead of an opaque `__missingCapabilities`. The
34
+ // key distributes over a union of missing names (one required key each).
35
+ [K in `✖ Missing capability "${TMissing}": no configured middleware provides it. Add a middleware whose \`provides\` includes it (and, with createChatMiddleware().use(), order the provider before this consumer).`]: never
36
+ }
37
+
38
+ /**
39
+ * Missing capability names. When required names are imprecise (`string`, i.e.
40
+ * plain `ChatMiddleware` not authored via `defineChatMiddleware`), we cannot
41
+ * prove a gap, so we allow it (→ `never`). Otherwise the precise literals not
42
+ * present in the provided set.
43
+ */
44
+ type MissingNames<TList extends ReadonlyArray<AnyChatMiddleware>> =
45
+ string extends RequiredNames<TList>
46
+ ? never
47
+ : Exclude<RequiredNames<TList>, ProvidedNames<TList>>
48
+
49
+ /**
50
+ * Resolves to `TList` when coverage holds, otherwise to a `MissingCapabilities`
51
+ * marker (not assignable to a middleware array) — producing a compile error at
52
+ * the `middleware` option that names the missing capability.
53
+ */
54
+ export type CheckCoverage<TList extends ReadonlyArray<AnyChatMiddleware>> = [
55
+ MissingNames<TList>,
56
+ ] extends [never]
57
+ ? TList
58
+ : MissingCapabilities<MissingNames<TList>>
59
+
60
+ /**
61
+ * Order-aware middleware builder. Each `.use()` requires that the middleware's
62
+ * required capability names are already in the accumulated provided set, then
63
+ * adds its provided names. `.build()` returns the ordered array.
64
+ *
65
+ * `TProvided` is the running union of provided capability name literals.
66
+ */
67
+ export interface ChatMiddlewareBuilder<
68
+ TList extends ReadonlyArray<AnyChatMiddleware>,
69
+ TProvided extends string,
70
+ > {
71
+ use: <
72
+ TRequires extends ReadonlyArray<CapabilityHandle>,
73
+ TProvides extends ReadonlyArray<CapabilityHandle>,
74
+ TContext = unknown,
75
+ >(
76
+ middleware: [NamesOf<TRequires>] extends [TProvided]
77
+ ? DefinedChatMiddleware<TContext, TRequires, TProvides>
78
+ : DefinedChatMiddleware<TContext, TRequires, TProvides> &
79
+ MissingCapabilities<Exclude<NamesOf<TRequires>, TProvided>>,
80
+ ) => ChatMiddlewareBuilder<
81
+ readonly [...TList, DefinedChatMiddleware<TContext, TRequires, TProvides>],
82
+ TProvided | NamesOf<TProvides>
83
+ >
84
+
85
+ build: () => [...TList]
86
+ }
87
+
88
+ /** Create an order-aware middleware builder. */
89
+ export function createChatMiddleware(): ChatMiddlewareBuilder<
90
+ readonly [],
91
+ never
92
+ > {
93
+ const list: Array<ChatMiddleware<unknown>> = []
94
+ const builder = {
95
+ use(middleware: ChatMiddleware<unknown>) {
96
+ list.push(middleware)
97
+ return builder
98
+ },
99
+ build() {
100
+ return list
101
+ },
102
+ }
103
+ // The only sanctioned assertion in this PR: the runtime `builder` is a single
104
+ // object reused across `.use()` calls, but the type accumulates `TProvided`
105
+ // and `TList` per call — TypeScript cannot derive that from runtime values, so
106
+ // a structural `as` is impossible and the double assertion is irreducible.
107
+ // eslint-disable-next-line no-restricted-syntax -- irreducible: type-level accumulation cannot be expressed from a single runtime object
108
+ return builder as unknown as ChatMiddlewareBuilder<readonly [], never>
109
+ }