@tanstack/ai-fal 0.8.2 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,246 @@
1
+ import { FAL_IMAGE_FIELD_OVERRIDES } from './generated/image-field-overrides'
2
+ import type {
3
+ FalImageFieldName,
4
+ FalImageFieldOverride,
5
+ } from './generated/image-field-overrides'
6
+ import type { ImagePart, MediaInputMetadata } from '@tanstack/ai'
7
+ import type { FalModel, FalModelInput } from '../model-meta'
8
+
9
+ /**
10
+ * The image-conditioning fields the mappers may set, narrowed to the ones
11
+ * that actually exist on the given endpoint's input type. For endpoints
12
+ * unknown to the installed `@fal-ai/client` this widens to all known field
13
+ * names.
14
+ */
15
+ export type FalImageInputFields<TModel extends string> = Partial<
16
+ Pick<
17
+ FalModelInput<TModel>,
18
+ Extract<keyof FalModelInput<TModel>, FalImageFieldName>
19
+ >
20
+ >
21
+
22
+ /**
23
+ * Default field per routing role. Endpoint-specific deviations live in the
24
+ * generated `FAL_IMAGE_FIELD_OVERRIDES` map (regenerate with
25
+ * `pnpm generate:fal-image-fields`); these defaults must stay in sync with
26
+ * `DEFAULTS` in scripts/generate-fal-image-field-map.ts.
27
+ */
28
+ const DEFAULT_FIELDS = {
29
+ single: 'image_url',
30
+ multi: 'image_urls',
31
+ mask: 'mask_url',
32
+ control: 'control_image_url',
33
+ reference: 'reference_image_urls',
34
+ start: 'start_image_url',
35
+ end: 'end_image_url',
36
+ } satisfies Required<FalImageFieldOverride>
37
+
38
+ /**
39
+ * Field names that accept an array of images. The generator asserts the
40
+ * SDK types agree with this set, so wrap-vs-scalar decisions stay correct.
41
+ */
42
+ const LIST_FIELDS = new Set<string>([
43
+ 'image_urls',
44
+ 'input_image_urls',
45
+ 'ref_image_urls',
46
+ 'reference_image_urls',
47
+ ])
48
+
49
+ /** Resolve the per-role field names for a model: defaults + generated overrides. */
50
+ function fieldSpecFor(model: string): Required<FalImageFieldOverride> {
51
+ const overrides = (
52
+ FAL_IMAGE_FIELD_OVERRIDES as Record<string, FalImageFieldOverride>
53
+ )[model]
54
+ return { ...DEFAULT_FIELDS, ...overrides }
55
+ }
56
+
57
+ /**
58
+ * Assign URLs to a field, wrapping or unwrapping based on whether the field
59
+ * takes an array. When two roles resolve to the same list field (e.g.
60
+ * sources and references both land on `image_urls` for nano-banana edit)
61
+ * the values are merged in assignment order; two roles resolving to the
62
+ * same scalar field is ambiguous and throws. Throws when multiple images
63
+ * target a scalar field.
64
+ */
65
+ function assignField(
66
+ fields: Record<string, unknown>,
67
+ field: string,
68
+ urls: Array<string>,
69
+ model: string,
70
+ what: string,
71
+ ): void {
72
+ if (urls.length === 0) return
73
+ const existing = fields[field]
74
+ if (LIST_FIELDS.has(field)) {
75
+ fields[field] = Array.isArray(existing) ? [...existing, ...urls] : urls
76
+ } else if (existing !== undefined) {
77
+ throw new Error(
78
+ `fal: multiple inputs map to '${field}' on model ${model}. Drop one of the conflicting inputs or pass the field explicitly via modelOptions.`,
79
+ )
80
+ } else if (urls.length === 1) {
81
+ fields[field] = urls[0]
82
+ } else {
83
+ throw new Error(
84
+ `fal: model ${model} accepts a single ${what} image via '${field}' (received ${urls.length}).`,
85
+ )
86
+ }
87
+ }
88
+
89
+ interface RoleBuckets {
90
+ sources: Array<string>
91
+ masks: Array<string>
92
+ controls: Array<string>
93
+ references: Array<string>
94
+ starts: Array<string>
95
+ ends: Array<string>
96
+ }
97
+
98
+ function bucketByRole(
99
+ imageInputs: ReadonlyArray<ImagePart<MediaInputMetadata>>,
100
+ ): RoleBuckets {
101
+ const buckets: RoleBuckets = {
102
+ sources: [],
103
+ masks: [],
104
+ controls: [],
105
+ references: [],
106
+ starts: [],
107
+ ends: [],
108
+ }
109
+ for (const part of imageInputs) {
110
+ const url = imagePartToUrl(part)
111
+ const role = part.metadata?.role
112
+ if (role === 'mask') buckets.masks.push(url)
113
+ else if (role === 'control') buckets.controls.push(url)
114
+ else if (role === 'reference' || role === 'character')
115
+ buckets.references.push(url)
116
+ else if (role === 'start_frame') buckets.starts.push(url)
117
+ else if (role === 'end_frame') buckets.ends.push(url)
118
+ else buckets.sources.push(url)
119
+ }
120
+ return buckets
121
+ }
122
+
123
+ /**
124
+ * Map the prompt's image parts onto fal.ai image-endpoint fields.
125
+ *
126
+ * fal endpoints use different field names for image-conditioned generation
127
+ * (~80% use `image_url` for single; the rest use `image_urls`,
128
+ * `reference_image_urls`, `mask_url`, `control_image_url`, etc.). Field
129
+ * names are resolved per endpoint from the generated
130
+ * `FAL_IMAGE_FIELD_OVERRIDES` map (derived from the fal SDK's endpoint
131
+ * types), falling back to the defaults above for endpoints the installed
132
+ * SDK doesn't know:
133
+ *
134
+ * - parts with `metadata.role === 'mask'` → spec.mask (single)
135
+ * - parts with `metadata.role === 'control'` → spec.control (single)
136
+ * - `role === 'reference' | 'character'` → spec.reference
137
+ * - `role === 'start_frame' | 'end_frame'` → treated as sources (frame
138
+ * roles only apply to video generation)
139
+ * - remaining parts → spec.single / spec.multi
140
+ *
141
+ * Users can always override the resulting field shape via `modelOptions`
142
+ * (spread before these fields), or pass everything through `modelOptions`
143
+ * directly when the mapping doesn't match an obscure endpoint.
144
+ */
145
+ export function mapImageInputsToFalFields<TModel extends FalModel>(
146
+ model: TModel,
147
+ imageInputs?: ReadonlyArray<ImagePart<MediaInputMetadata>>,
148
+ ): FalImageInputFields<TModel> {
149
+ if (!imageInputs || imageInputs.length === 0) return {}
150
+
151
+ const spec = fieldSpecFor(model)
152
+ const { sources, masks, controls, references, starts, ends } =
153
+ bucketByRole(imageInputs)
154
+ // Frame roles aren't meaningful for image generation; treat as the
155
+ // primary source. The video mapper handles start/end framing.
156
+ const allSources = [...sources, ...starts, ...ends]
157
+
158
+ if (masks.length > 1) {
159
+ throw new Error(
160
+ `fal: only one input with metadata.role === 'mask' is supported per request (received ${masks.length}).`,
161
+ )
162
+ }
163
+ if (controls.length > 1) {
164
+ throw new Error(
165
+ `fal: only one input with metadata.role === 'control' is supported per request (received ${controls.length}).`,
166
+ )
167
+ }
168
+
169
+ const fields: Record<string, unknown> = {}
170
+ const sourceField = allSources.length > 1 ? spec.multi : spec.single
171
+ assignField(fields, sourceField, allSources, model, 'source')
172
+ assignField(fields, spec.reference, references, model, 'reference')
173
+ assignField(fields, spec.mask, masks, model, 'mask')
174
+ assignField(fields, spec.control, controls, model, 'control')
175
+
176
+ return fields as FalImageInputFields<TModel>
177
+ }
178
+
179
+ /**
180
+ * Map the prompt's image parts onto fal.ai video-endpoint fields.
181
+ *
182
+ * Video endpoints often expose a start frame as `image_url` (76% of i2v
183
+ * models) plus an optional `end_image_url`. Multi-reference video models
184
+ * (Kling O3, Seedance reference-to-video) use `reference_image_urls` or
185
+ * `image_urls`. Field names resolve through the same generated override
186
+ * map as the image mapper — e.g. `role: 'start_frame'` lands on `image_url`
187
+ * for Kling/Veo image-to-video and `first_frame_url` for Pixverse. Mapping:
188
+ *
189
+ * - `metadata.role === 'start_frame'` → spec.start
190
+ * - `metadata.role === 'end_frame'` → spec.end
191
+ * - `metadata.role === 'reference' | 'character'` → spec.reference
192
+ * - `metadata.role === 'mask' | 'control'` → throws (no video routing)
193
+ * - remaining parts (no role) → spec.single / spec.multi
194
+ */
195
+ export function mapImageInputsToFalVideoFields<TModel extends FalModel>(
196
+ model: TModel,
197
+ imageInputs?: ReadonlyArray<ImagePart<MediaInputMetadata>>,
198
+ ): FalImageInputFields<TModel> {
199
+ if (!imageInputs || imageInputs.length === 0) return {}
200
+
201
+ const spec = fieldSpecFor(model)
202
+ const { sources, masks, controls, references, starts, ends } =
203
+ bucketByRole(imageInputs)
204
+ // Mask / control roles have no video-specific routing; silently repurposing
205
+ // them as source frames would hide the problem, so reject them instead.
206
+ if (masks.length > 0 || controls.length > 0) {
207
+ const role = masks.length > 0 ? 'mask' : 'control'
208
+ throw new Error(
209
+ `fal: metadata.role === '${role}' is not supported for video generation on model ${model}. ` +
210
+ `Remove the role or pass the field explicitly via modelOptions.`,
211
+ )
212
+ }
213
+
214
+ if (starts.length > 1) {
215
+ throw new Error(
216
+ `fal: only one input with metadata.role === 'start_frame' is supported (received ${starts.length}).`,
217
+ )
218
+ }
219
+ if (ends.length > 1) {
220
+ throw new Error(
221
+ `fal: only one input with metadata.role === 'end_frame' is supported (received ${ends.length}).`,
222
+ )
223
+ }
224
+
225
+ const fields: Record<string, unknown> = {}
226
+ const sourceField = sources.length > 1 ? spec.multi : spec.single
227
+ assignField(fields, sourceField, sources, model, 'source')
228
+ assignField(fields, spec.reference, references, model, 'reference')
229
+ // Frame roles assign last: when an endpoint routes the start frame to its
230
+ // generic source field (e.g. Kling image-to-video) and an unroled source
231
+ // was also provided, assignField rejects the ambiguous combination.
232
+ assignField(fields, spec.start, starts, model, 'start frame')
233
+ assignField(fields, spec.end, ends, model, 'end frame')
234
+
235
+ return fields as FalImageInputFields<TModel>
236
+ }
237
+
238
+ /**
239
+ * Convert a TanStack ImagePart into a string suitable for fal's URL-based
240
+ * input fields. URL sources pass through; data sources are emitted as a
241
+ * `data:<mime>;base64,<value>` URI which fal endpoints accept on the wire.
242
+ */
243
+ function imagePartToUrl(part: ImagePart<MediaInputMetadata>): string {
244
+ if (part.source.type === 'url') return part.source.value
245
+ return `data:${part.source.mimeType};base64,${part.source.value}`
246
+ }
package/src/model-meta.ts CHANGED
@@ -4,6 +4,8 @@
4
4
  * These types give you full autocomplete and type safety for any model.
5
5
  */
6
6
  import type { EndpointTypeMap } from '@fal-ai/client/endpoints'
7
+ import type { MediaPromptModality } from '@tanstack/ai'
8
+ import type { FalImageFieldName } from './image/generated/image-field-overrides'
7
9
 
8
10
  export type { EndpointTypeMap } from '@fal-ai/client/endpoints'
9
11
 
@@ -70,6 +72,32 @@ export type FalModelImageSizeInput<TModel extends string> =
70
72
  : never
71
73
  : { image_size: string }
72
74
 
75
+ /**
76
+ * Input fields the prompt-part mappers can populate: image conditioning via
77
+ * the generated `FalImageFieldName` set, video conditioning via
78
+ * `video_url` / `video_urls` / `reference_video_urls`, audio via `audio_url`.
79
+ */
80
+ type FalMediaInputFieldName =
81
+ | FalImageFieldName
82
+ | 'video_url'
83
+ | 'video_urls'
84
+ | 'reference_video_urls'
85
+ | 'audio_url'
86
+
87
+ /**
88
+ * Demote an endpoint input's media-conditioning fields from required to
89
+ * optional. Image-to-video endpoints declare e.g. `image_url` as a required
90
+ * input, but with a multimodal `prompt` the start frame usually arrives as a
91
+ * prompt part — requiring it in `modelOptions` too would force redundancy.
92
+ * The fields stay passable via `modelOptions` as the documented escape hatch
93
+ * (and override-wise the mapped prompt-part fields win on conflict).
94
+ */
95
+ type WithOptionalMediaInputFields<TInput> = Omit<
96
+ TInput,
97
+ Extract<keyof TInput, FalMediaInputFieldName>
98
+ > &
99
+ Partial<Pick<TInput, Extract<keyof TInput, FalMediaInputFieldName>>>
100
+
73
101
  /**
74
102
  * Provider options for image generation, excluding fields TanStack AI handles.
75
103
  * Use this for the `modelOptions` parameter in image generation.
@@ -78,10 +106,8 @@ export type FalModelImageSizeInput<TModel extends string> =
78
106
  * type FluxOptions = FalImageProviderOptions<'fal-ai/flux/dev'>
79
107
  * // { num_inference_steps?: number; guidance_scale?: number; seed?: number; ... }
80
108
  */
81
- export type FalImageProviderOptions<TModel extends string> = Omit<
82
- FalModelInput<TModel>,
83
- 'prompt'
84
- >
109
+ export type FalImageProviderOptions<TModel extends string> =
110
+ WithOptionalMediaInputFields<Omit<FalModelInput<TModel>, 'prompt'>>
85
111
 
86
112
  /**
87
113
  * Extract the video size type supported by a specific fal model.
@@ -118,13 +144,57 @@ export type FalModelVideoSizeInput<TModel extends string> =
118
144
  : never
119
145
  : { aspect_ratio?: string; resolution?: string }
120
146
 
147
+ /**
148
+ * Prompt input modalities for a fal image endpoint, derived from the SDK's
149
+ * endpoint input type: an endpoint accepts image prompt parts exactly when
150
+ * its input declares one of the known image-conditioning fields
151
+ * (`image_url`, `image_urls`, `mask_url`, …). Endpoints unknown to the
152
+ * installed SDK are unconstrained.
153
+ */
154
+ export type FalImagePromptModalitiesFor<TModel extends string> =
155
+ TModel extends keyof EndpointTypeMap
156
+ ? ReadonlyArray<
157
+ Extract<keyof FalModelInput<TModel>, FalImageFieldName> extends never
158
+ ? never
159
+ : 'image'
160
+ >
161
+ : ReadonlyArray<MediaPromptModality>
162
+
163
+ /**
164
+ * Prompt input modalities for a fal video endpoint. Image conditioning is
165
+ * detected via the same field set as image endpoints; video conditioning via
166
+ * `video_url` / `video_urls` / `reference_video_urls`; audio conditioning
167
+ * via `audio_url`. Endpoints unknown to the installed SDK are unconstrained.
168
+ */
169
+ export type FalVideoPromptModalitiesFor<TModel extends string> =
170
+ TModel extends keyof EndpointTypeMap
171
+ ? ReadonlyArray<
172
+ | (Extract<keyof FalModelInput<TModel>, FalImageFieldName> extends never
173
+ ? never
174
+ : 'image')
175
+ | (Extract<
176
+ keyof FalModelInput<TModel>,
177
+ 'video_url' | 'video_urls' | 'reference_video_urls'
178
+ > extends never
179
+ ? never
180
+ : 'video')
181
+ | (Extract<keyof FalModelInput<TModel>, 'audio_url'> extends never
182
+ ? never
183
+ : 'audio')
184
+ >
185
+ : ReadonlyArray<MediaPromptModality>
186
+
121
187
  /**
122
188
  * Provider options for video generation, excluding fields TanStack AI handles.
123
189
  * Use this for the `modelOptions` parameter in video generation.
190
+ *
191
+ * Media-conditioning fields (start/end frame, reference images, source
192
+ * video/audio) are optional here even when the endpoint requires them —
193
+ * they're usually supplied as prompt parts instead.
124
194
  */
125
195
  export type FalVideoProviderOptions<TModel extends string> =
126
196
  TModel extends keyof EndpointTypeMap
127
- ? Omit<FalModelInput<TModel>, 'prompt'>
197
+ ? WithOptionalMediaInputFields<Omit<FalModelInput<TModel>, 'prompt'>>
128
198
  : Record<string, unknown>
129
199
 
130
200
  /**