@tanstack/ai-fal 0.8.2 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.d.ts +2 -2
- package/dist/esm/adapters/image.js +21 -4
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +2 -2
- package/dist/esm/adapters/video.js +55 -3
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/image/generated/image-field-overrides.d.ts +1279 -0
- package/dist/esm/image/generated/image-field-overrides.js +368 -0
- package/dist/esm/image/generated/image-field-overrides.js.map +1 -0
- package/dist/esm/image/image-inputs.d.ts +50 -0
- package/dist/esm/image/image-inputs.js +119 -0
- package/dist/esm/image/image-inputs.js.map +1 -0
- package/dist/esm/model-meta.d.ts +38 -2
- package/package.json +3 -3
- package/src/adapters/image.ts +31 -4
- package/src/adapters/video.ts +82 -4
- package/src/image/generated/image-field-overrides.ts +430 -0
- package/src/image/image-inputs.ts +246 -0
- package/src/model-meta.ts +75 -5
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
import { FAL_IMAGE_FIELD_OVERRIDES } from './generated/image-field-overrides'
|
|
2
|
+
import type {
|
|
3
|
+
FalImageFieldName,
|
|
4
|
+
FalImageFieldOverride,
|
|
5
|
+
} from './generated/image-field-overrides'
|
|
6
|
+
import type { ImagePart, MediaInputMetadata } from '@tanstack/ai'
|
|
7
|
+
import type { FalModel, FalModelInput } from '../model-meta'
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* The image-conditioning fields the mappers may set, narrowed to the ones
|
|
11
|
+
* that actually exist on the given endpoint's input type. For endpoints
|
|
12
|
+
* unknown to the installed `@fal-ai/client` this widens to all known field
|
|
13
|
+
* names.
|
|
14
|
+
*/
|
|
15
|
+
export type FalImageInputFields<TModel extends string> = Partial<
|
|
16
|
+
Pick<
|
|
17
|
+
FalModelInput<TModel>,
|
|
18
|
+
Extract<keyof FalModelInput<TModel>, FalImageFieldName>
|
|
19
|
+
>
|
|
20
|
+
>
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Default field per routing role. Endpoint-specific deviations live in the
|
|
24
|
+
* generated `FAL_IMAGE_FIELD_OVERRIDES` map (regenerate with
|
|
25
|
+
* `pnpm generate:fal-image-fields`); these defaults must stay in sync with
|
|
26
|
+
* `DEFAULTS` in scripts/generate-fal-image-field-map.ts.
|
|
27
|
+
*/
|
|
28
|
+
const DEFAULT_FIELDS = {
|
|
29
|
+
single: 'image_url',
|
|
30
|
+
multi: 'image_urls',
|
|
31
|
+
mask: 'mask_url',
|
|
32
|
+
control: 'control_image_url',
|
|
33
|
+
reference: 'reference_image_urls',
|
|
34
|
+
start: 'start_image_url',
|
|
35
|
+
end: 'end_image_url',
|
|
36
|
+
} satisfies Required<FalImageFieldOverride>
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Field names that accept an array of images. The generator asserts the
|
|
40
|
+
* SDK types agree with this set, so wrap-vs-scalar decisions stay correct.
|
|
41
|
+
*/
|
|
42
|
+
const LIST_FIELDS = new Set<string>([
|
|
43
|
+
'image_urls',
|
|
44
|
+
'input_image_urls',
|
|
45
|
+
'ref_image_urls',
|
|
46
|
+
'reference_image_urls',
|
|
47
|
+
])
|
|
48
|
+
|
|
49
|
+
/** Resolve the per-role field names for a model: defaults + generated overrides. */
|
|
50
|
+
function fieldSpecFor(model: string): Required<FalImageFieldOverride> {
|
|
51
|
+
const overrides = (
|
|
52
|
+
FAL_IMAGE_FIELD_OVERRIDES as Record<string, FalImageFieldOverride>
|
|
53
|
+
)[model]
|
|
54
|
+
return { ...DEFAULT_FIELDS, ...overrides }
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Assign URLs to a field, wrapping or unwrapping based on whether the field
|
|
59
|
+
* takes an array. When two roles resolve to the same list field (e.g.
|
|
60
|
+
* sources and references both land on `image_urls` for nano-banana edit)
|
|
61
|
+
* the values are merged in assignment order; two roles resolving to the
|
|
62
|
+
* same scalar field is ambiguous and throws. Throws when multiple images
|
|
63
|
+
* target a scalar field.
|
|
64
|
+
*/
|
|
65
|
+
function assignField(
|
|
66
|
+
fields: Record<string, unknown>,
|
|
67
|
+
field: string,
|
|
68
|
+
urls: Array<string>,
|
|
69
|
+
model: string,
|
|
70
|
+
what: string,
|
|
71
|
+
): void {
|
|
72
|
+
if (urls.length === 0) return
|
|
73
|
+
const existing = fields[field]
|
|
74
|
+
if (LIST_FIELDS.has(field)) {
|
|
75
|
+
fields[field] = Array.isArray(existing) ? [...existing, ...urls] : urls
|
|
76
|
+
} else if (existing !== undefined) {
|
|
77
|
+
throw new Error(
|
|
78
|
+
`fal: multiple inputs map to '${field}' on model ${model}. Drop one of the conflicting inputs or pass the field explicitly via modelOptions.`,
|
|
79
|
+
)
|
|
80
|
+
} else if (urls.length === 1) {
|
|
81
|
+
fields[field] = urls[0]
|
|
82
|
+
} else {
|
|
83
|
+
throw new Error(
|
|
84
|
+
`fal: model ${model} accepts a single ${what} image via '${field}' (received ${urls.length}).`,
|
|
85
|
+
)
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
interface RoleBuckets {
|
|
90
|
+
sources: Array<string>
|
|
91
|
+
masks: Array<string>
|
|
92
|
+
controls: Array<string>
|
|
93
|
+
references: Array<string>
|
|
94
|
+
starts: Array<string>
|
|
95
|
+
ends: Array<string>
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
function bucketByRole(
|
|
99
|
+
imageInputs: ReadonlyArray<ImagePart<MediaInputMetadata>>,
|
|
100
|
+
): RoleBuckets {
|
|
101
|
+
const buckets: RoleBuckets = {
|
|
102
|
+
sources: [],
|
|
103
|
+
masks: [],
|
|
104
|
+
controls: [],
|
|
105
|
+
references: [],
|
|
106
|
+
starts: [],
|
|
107
|
+
ends: [],
|
|
108
|
+
}
|
|
109
|
+
for (const part of imageInputs) {
|
|
110
|
+
const url = imagePartToUrl(part)
|
|
111
|
+
const role = part.metadata?.role
|
|
112
|
+
if (role === 'mask') buckets.masks.push(url)
|
|
113
|
+
else if (role === 'control') buckets.controls.push(url)
|
|
114
|
+
else if (role === 'reference' || role === 'character')
|
|
115
|
+
buckets.references.push(url)
|
|
116
|
+
else if (role === 'start_frame') buckets.starts.push(url)
|
|
117
|
+
else if (role === 'end_frame') buckets.ends.push(url)
|
|
118
|
+
else buckets.sources.push(url)
|
|
119
|
+
}
|
|
120
|
+
return buckets
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Map the prompt's image parts onto fal.ai image-endpoint fields.
|
|
125
|
+
*
|
|
126
|
+
* fal endpoints use different field names for image-conditioned generation
|
|
127
|
+
* (~80% use `image_url` for single; the rest use `image_urls`,
|
|
128
|
+
* `reference_image_urls`, `mask_url`, `control_image_url`, etc.). Field
|
|
129
|
+
* names are resolved per endpoint from the generated
|
|
130
|
+
* `FAL_IMAGE_FIELD_OVERRIDES` map (derived from the fal SDK's endpoint
|
|
131
|
+
* types), falling back to the defaults above for endpoints the installed
|
|
132
|
+
* SDK doesn't know:
|
|
133
|
+
*
|
|
134
|
+
* - parts with `metadata.role === 'mask'` → spec.mask (single)
|
|
135
|
+
* - parts with `metadata.role === 'control'` → spec.control (single)
|
|
136
|
+
* - `role === 'reference' | 'character'` → spec.reference
|
|
137
|
+
* - `role === 'start_frame' | 'end_frame'` → treated as sources (frame
|
|
138
|
+
* roles only apply to video generation)
|
|
139
|
+
* - remaining parts → spec.single / spec.multi
|
|
140
|
+
*
|
|
141
|
+
* Users can always override the resulting field shape via `modelOptions`
|
|
142
|
+
* (spread before these fields), or pass everything through `modelOptions`
|
|
143
|
+
* directly when the mapping doesn't match an obscure endpoint.
|
|
144
|
+
*/
|
|
145
|
+
export function mapImageInputsToFalFields<TModel extends FalModel>(
|
|
146
|
+
model: TModel,
|
|
147
|
+
imageInputs?: ReadonlyArray<ImagePart<MediaInputMetadata>>,
|
|
148
|
+
): FalImageInputFields<TModel> {
|
|
149
|
+
if (!imageInputs || imageInputs.length === 0) return {}
|
|
150
|
+
|
|
151
|
+
const spec = fieldSpecFor(model)
|
|
152
|
+
const { sources, masks, controls, references, starts, ends } =
|
|
153
|
+
bucketByRole(imageInputs)
|
|
154
|
+
// Frame roles aren't meaningful for image generation; treat as the
|
|
155
|
+
// primary source. The video mapper handles start/end framing.
|
|
156
|
+
const allSources = [...sources, ...starts, ...ends]
|
|
157
|
+
|
|
158
|
+
if (masks.length > 1) {
|
|
159
|
+
throw new Error(
|
|
160
|
+
`fal: only one input with metadata.role === 'mask' is supported per request (received ${masks.length}).`,
|
|
161
|
+
)
|
|
162
|
+
}
|
|
163
|
+
if (controls.length > 1) {
|
|
164
|
+
throw new Error(
|
|
165
|
+
`fal: only one input with metadata.role === 'control' is supported per request (received ${controls.length}).`,
|
|
166
|
+
)
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
const fields: Record<string, unknown> = {}
|
|
170
|
+
const sourceField = allSources.length > 1 ? spec.multi : spec.single
|
|
171
|
+
assignField(fields, sourceField, allSources, model, 'source')
|
|
172
|
+
assignField(fields, spec.reference, references, model, 'reference')
|
|
173
|
+
assignField(fields, spec.mask, masks, model, 'mask')
|
|
174
|
+
assignField(fields, spec.control, controls, model, 'control')
|
|
175
|
+
|
|
176
|
+
return fields as FalImageInputFields<TModel>
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* Map the prompt's image parts onto fal.ai video-endpoint fields.
|
|
181
|
+
*
|
|
182
|
+
* Video endpoints often expose a start frame as `image_url` (76% of i2v
|
|
183
|
+
* models) plus an optional `end_image_url`. Multi-reference video models
|
|
184
|
+
* (Kling O3, Seedance reference-to-video) use `reference_image_urls` or
|
|
185
|
+
* `image_urls`. Field names resolve through the same generated override
|
|
186
|
+
* map as the image mapper — e.g. `role: 'start_frame'` lands on `image_url`
|
|
187
|
+
* for Kling/Veo image-to-video and `first_frame_url` for Pixverse. Mapping:
|
|
188
|
+
*
|
|
189
|
+
* - `metadata.role === 'start_frame'` → spec.start
|
|
190
|
+
* - `metadata.role === 'end_frame'` → spec.end
|
|
191
|
+
* - `metadata.role === 'reference' | 'character'` → spec.reference
|
|
192
|
+
* - `metadata.role === 'mask' | 'control'` → throws (no video routing)
|
|
193
|
+
* - remaining parts (no role) → spec.single / spec.multi
|
|
194
|
+
*/
|
|
195
|
+
export function mapImageInputsToFalVideoFields<TModel extends FalModel>(
|
|
196
|
+
model: TModel,
|
|
197
|
+
imageInputs?: ReadonlyArray<ImagePart<MediaInputMetadata>>,
|
|
198
|
+
): FalImageInputFields<TModel> {
|
|
199
|
+
if (!imageInputs || imageInputs.length === 0) return {}
|
|
200
|
+
|
|
201
|
+
const spec = fieldSpecFor(model)
|
|
202
|
+
const { sources, masks, controls, references, starts, ends } =
|
|
203
|
+
bucketByRole(imageInputs)
|
|
204
|
+
// Mask / control roles have no video-specific routing; silently repurposing
|
|
205
|
+
// them as source frames would hide the problem, so reject them instead.
|
|
206
|
+
if (masks.length > 0 || controls.length > 0) {
|
|
207
|
+
const role = masks.length > 0 ? 'mask' : 'control'
|
|
208
|
+
throw new Error(
|
|
209
|
+
`fal: metadata.role === '${role}' is not supported for video generation on model ${model}. ` +
|
|
210
|
+
`Remove the role or pass the field explicitly via modelOptions.`,
|
|
211
|
+
)
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
if (starts.length > 1) {
|
|
215
|
+
throw new Error(
|
|
216
|
+
`fal: only one input with metadata.role === 'start_frame' is supported (received ${starts.length}).`,
|
|
217
|
+
)
|
|
218
|
+
}
|
|
219
|
+
if (ends.length > 1) {
|
|
220
|
+
throw new Error(
|
|
221
|
+
`fal: only one input with metadata.role === 'end_frame' is supported (received ${ends.length}).`,
|
|
222
|
+
)
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
const fields: Record<string, unknown> = {}
|
|
226
|
+
const sourceField = sources.length > 1 ? spec.multi : spec.single
|
|
227
|
+
assignField(fields, sourceField, sources, model, 'source')
|
|
228
|
+
assignField(fields, spec.reference, references, model, 'reference')
|
|
229
|
+
// Frame roles assign last: when an endpoint routes the start frame to its
|
|
230
|
+
// generic source field (e.g. Kling image-to-video) and an unroled source
|
|
231
|
+
// was also provided, assignField rejects the ambiguous combination.
|
|
232
|
+
assignField(fields, spec.start, starts, model, 'start frame')
|
|
233
|
+
assignField(fields, spec.end, ends, model, 'end frame')
|
|
234
|
+
|
|
235
|
+
return fields as FalImageInputFields<TModel>
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/**
|
|
239
|
+
* Convert a TanStack ImagePart into a string suitable for fal's URL-based
|
|
240
|
+
* input fields. URL sources pass through; data sources are emitted as a
|
|
241
|
+
* `data:<mime>;base64,<value>` URI which fal endpoints accept on the wire.
|
|
242
|
+
*/
|
|
243
|
+
function imagePartToUrl(part: ImagePart<MediaInputMetadata>): string {
|
|
244
|
+
if (part.source.type === 'url') return part.source.value
|
|
245
|
+
return `data:${part.source.mimeType};base64,${part.source.value}`
|
|
246
|
+
}
|
package/src/model-meta.ts
CHANGED
|
@@ -4,6 +4,8 @@
|
|
|
4
4
|
* These types give you full autocomplete and type safety for any model.
|
|
5
5
|
*/
|
|
6
6
|
import type { EndpointTypeMap } from '@fal-ai/client/endpoints'
|
|
7
|
+
import type { MediaPromptModality } from '@tanstack/ai'
|
|
8
|
+
import type { FalImageFieldName } from './image/generated/image-field-overrides'
|
|
7
9
|
|
|
8
10
|
export type { EndpointTypeMap } from '@fal-ai/client/endpoints'
|
|
9
11
|
|
|
@@ -70,6 +72,32 @@ export type FalModelImageSizeInput<TModel extends string> =
|
|
|
70
72
|
: never
|
|
71
73
|
: { image_size: string }
|
|
72
74
|
|
|
75
|
+
/**
|
|
76
|
+
* Input fields the prompt-part mappers can populate: image conditioning via
|
|
77
|
+
* the generated `FalImageFieldName` set, video conditioning via
|
|
78
|
+
* `video_url` / `video_urls` / `reference_video_urls`, audio via `audio_url`.
|
|
79
|
+
*/
|
|
80
|
+
type FalMediaInputFieldName =
|
|
81
|
+
| FalImageFieldName
|
|
82
|
+
| 'video_url'
|
|
83
|
+
| 'video_urls'
|
|
84
|
+
| 'reference_video_urls'
|
|
85
|
+
| 'audio_url'
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Demote an endpoint input's media-conditioning fields from required to
|
|
89
|
+
* optional. Image-to-video endpoints declare e.g. `image_url` as a required
|
|
90
|
+
* input, but with a multimodal `prompt` the start frame usually arrives as a
|
|
91
|
+
* prompt part — requiring it in `modelOptions` too would force redundancy.
|
|
92
|
+
* The fields stay passable via `modelOptions` as the documented escape hatch
|
|
93
|
+
* (and override-wise the mapped prompt-part fields win on conflict).
|
|
94
|
+
*/
|
|
95
|
+
type WithOptionalMediaInputFields<TInput> = Omit<
|
|
96
|
+
TInput,
|
|
97
|
+
Extract<keyof TInput, FalMediaInputFieldName>
|
|
98
|
+
> &
|
|
99
|
+
Partial<Pick<TInput, Extract<keyof TInput, FalMediaInputFieldName>>>
|
|
100
|
+
|
|
73
101
|
/**
|
|
74
102
|
* Provider options for image generation, excluding fields TanStack AI handles.
|
|
75
103
|
* Use this for the `modelOptions` parameter in image generation.
|
|
@@ -78,10 +106,8 @@ export type FalModelImageSizeInput<TModel extends string> =
|
|
|
78
106
|
* type FluxOptions = FalImageProviderOptions<'fal-ai/flux/dev'>
|
|
79
107
|
* // { num_inference_steps?: number; guidance_scale?: number; seed?: number; ... }
|
|
80
108
|
*/
|
|
81
|
-
export type FalImageProviderOptions<TModel extends string> =
|
|
82
|
-
FalModelInput<TModel>,
|
|
83
|
-
'prompt'
|
|
84
|
-
>
|
|
109
|
+
export type FalImageProviderOptions<TModel extends string> =
|
|
110
|
+
WithOptionalMediaInputFields<Omit<FalModelInput<TModel>, 'prompt'>>
|
|
85
111
|
|
|
86
112
|
/**
|
|
87
113
|
* Extract the video size type supported by a specific fal model.
|
|
@@ -118,13 +144,57 @@ export type FalModelVideoSizeInput<TModel extends string> =
|
|
|
118
144
|
: never
|
|
119
145
|
: { aspect_ratio?: string; resolution?: string }
|
|
120
146
|
|
|
147
|
+
/**
|
|
148
|
+
* Prompt input modalities for a fal image endpoint, derived from the SDK's
|
|
149
|
+
* endpoint input type: an endpoint accepts image prompt parts exactly when
|
|
150
|
+
* its input declares one of the known image-conditioning fields
|
|
151
|
+
* (`image_url`, `image_urls`, `mask_url`, …). Endpoints unknown to the
|
|
152
|
+
* installed SDK are unconstrained.
|
|
153
|
+
*/
|
|
154
|
+
export type FalImagePromptModalitiesFor<TModel extends string> =
|
|
155
|
+
TModel extends keyof EndpointTypeMap
|
|
156
|
+
? ReadonlyArray<
|
|
157
|
+
Extract<keyof FalModelInput<TModel>, FalImageFieldName> extends never
|
|
158
|
+
? never
|
|
159
|
+
: 'image'
|
|
160
|
+
>
|
|
161
|
+
: ReadonlyArray<MediaPromptModality>
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Prompt input modalities for a fal video endpoint. Image conditioning is
|
|
165
|
+
* detected via the same field set as image endpoints; video conditioning via
|
|
166
|
+
* `video_url` / `video_urls` / `reference_video_urls`; audio conditioning
|
|
167
|
+
* via `audio_url`. Endpoints unknown to the installed SDK are unconstrained.
|
|
168
|
+
*/
|
|
169
|
+
export type FalVideoPromptModalitiesFor<TModel extends string> =
|
|
170
|
+
TModel extends keyof EndpointTypeMap
|
|
171
|
+
? ReadonlyArray<
|
|
172
|
+
| (Extract<keyof FalModelInput<TModel>, FalImageFieldName> extends never
|
|
173
|
+
? never
|
|
174
|
+
: 'image')
|
|
175
|
+
| (Extract<
|
|
176
|
+
keyof FalModelInput<TModel>,
|
|
177
|
+
'video_url' | 'video_urls' | 'reference_video_urls'
|
|
178
|
+
> extends never
|
|
179
|
+
? never
|
|
180
|
+
: 'video')
|
|
181
|
+
| (Extract<keyof FalModelInput<TModel>, 'audio_url'> extends never
|
|
182
|
+
? never
|
|
183
|
+
: 'audio')
|
|
184
|
+
>
|
|
185
|
+
: ReadonlyArray<MediaPromptModality>
|
|
186
|
+
|
|
121
187
|
/**
|
|
122
188
|
* Provider options for video generation, excluding fields TanStack AI handles.
|
|
123
189
|
* Use this for the `modelOptions` parameter in video generation.
|
|
190
|
+
*
|
|
191
|
+
* Media-conditioning fields (start/end frame, reference images, source
|
|
192
|
+
* video/audio) are optional here even when the endpoint requires them —
|
|
193
|
+
* they're usually supplied as prompt parts instead.
|
|
124
194
|
*/
|
|
125
195
|
export type FalVideoProviderOptions<TModel extends string> =
|
|
126
196
|
TModel extends keyof EndpointTypeMap
|
|
127
|
-
? Omit<FalModelInput<TModel>, 'prompt'
|
|
197
|
+
? WithOptionalMediaInputFields<Omit<FalModelInput<TModel>, 'prompt'>>
|
|
128
198
|
: Record<string, unknown>
|
|
129
199
|
|
|
130
200
|
/**
|