@tanstack/ai-grok 0.14.10 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.js +1 -1
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/tts.js +2 -1
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +33 -11
- package/dist/esm/adapters/video.js +125 -27
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/image/image-provider-options.d.ts +17 -2
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +3 -3
- package/dist/esm/index.js +2 -2
- package/dist/esm/model-meta.d.ts +49 -3
- package/dist/esm/model-meta.js +110 -8
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.js +17 -16
- package/dist/esm/realtime/adapter.js.map +1 -1
- package/dist/esm/realtime/token.d.ts +1 -1
- package/dist/esm/realtime/token.js +3 -2
- package/dist/esm/realtime/token.js.map +1 -1
- package/dist/esm/realtime/types.d.ts +1 -1
- package/dist/esm/tools/index.js.map +1 -1
- package/dist/esm/video/video-provider-options.d.ts +131 -21
- package/dist/esm/video/video-provider-options.js +36 -10
- package/dist/esm/video/video-provider-options.js.map +1 -1
- package/package.json +7 -7
- package/src/adapters/image.ts +2 -1
- package/src/adapters/video.ts +316 -51
- package/src/image/image-provider-options.ts +18 -2
- package/src/index.ts +7 -0
- package/src/model-meta.ts +109 -6
- package/src/realtime/adapter.ts +3 -2
- package/src/realtime/token.ts +4 -2
- package/src/realtime/types.ts +1 -1
- package/src/video/video-provider-options.ts +198 -34
package/src/adapters/video.ts
CHANGED
|
@@ -3,8 +3,11 @@ import { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'
|
|
|
3
3
|
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
4
4
|
import { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'
|
|
5
5
|
import {
|
|
6
|
+
GROK_VIDEO_MAX_REFERENCE_AUDIOS,
|
|
7
|
+
GROK_VIDEO_MAX_REFERENCE_IMAGES,
|
|
6
8
|
getGrokVideoDurationOptions,
|
|
7
|
-
|
|
9
|
+
isGrokVideoReferenceModel,
|
|
10
|
+
isGrokVideoSourceModel,
|
|
8
11
|
parseGrokVideoSize,
|
|
9
12
|
validateVideoSize,
|
|
10
13
|
} from '../video/video-provider-options'
|
|
@@ -15,6 +18,7 @@ import type {
|
|
|
15
18
|
TokenUsage,
|
|
16
19
|
VideoGenerationOptions,
|
|
17
20
|
VideoJobResult,
|
|
21
|
+
VideoPart,
|
|
18
22
|
VideoStatusResult,
|
|
19
23
|
VideoUrlResult,
|
|
20
24
|
} from '@tanstack/ai'
|
|
@@ -24,7 +28,7 @@ import type {
|
|
|
24
28
|
GrokVideoModelInputModalitiesByName,
|
|
25
29
|
GrokVideoModelProviderOptionsByName,
|
|
26
30
|
GrokVideoModelSizeByName,
|
|
27
|
-
|
|
31
|
+
GrokVideoRuntimeOptions,
|
|
28
32
|
} from '../video/video-provider-options'
|
|
29
33
|
import type { GrokClientConfig } from '../utils/client'
|
|
30
34
|
|
|
@@ -41,7 +45,7 @@ export interface GrokVideoConfig extends GrokClientConfig {}
|
|
|
41
45
|
*/
|
|
42
46
|
const USD_TICKS_PER_DOLLAR = 10_000_000_000
|
|
43
47
|
|
|
44
|
-
/** Response of POST /v1/videos/generations. */
|
|
48
|
+
/** Response of the POST /v1/videos/{generations,edits,extensions} endpoints. */
|
|
45
49
|
interface GrokVideoCreateResponse {
|
|
46
50
|
request_id?: string
|
|
47
51
|
}
|
|
@@ -62,11 +66,13 @@ interface GrokVideoStatusResponse {
|
|
|
62
66
|
}
|
|
63
67
|
|
|
64
68
|
/**
|
|
65
|
-
* Convert a TanStack
|
|
66
|
-
* video
|
|
67
|
-
* sources become base64 data URIs.
|
|
69
|
+
* Convert a TanStack image / video part to the URL string accepted by xAI's
|
|
70
|
+
* Imagine video endpoints: public URLs pass through (fetched by xAI's
|
|
71
|
+
* servers), data sources become base64 data URIs.
|
|
68
72
|
*/
|
|
69
|
-
function
|
|
73
|
+
function mediaPartToUrl(
|
|
74
|
+
part: ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata>,
|
|
75
|
+
): string {
|
|
70
76
|
if (part.source.type === 'url') return part.source.value
|
|
71
77
|
return `data:${part.source.mimeType};base64,${part.source.value}`
|
|
72
78
|
}
|
|
@@ -93,10 +99,10 @@ function buildGrokVideoUsage(
|
|
|
93
99
|
* async jobs/polling architecture: create a generation request, poll it,
|
|
94
100
|
* then read the completed video URL.
|
|
95
101
|
*
|
|
96
|
-
*
|
|
97
|
-
* `grok-imagine-video-1.5` is
|
|
98
|
-
*
|
|
99
|
-
*
|
|
102
|
+
* Both models support text-to-video and image-to-video;
|
|
103
|
+
* `grok-imagine-video-1.5` is xAI's documented default and adds native
|
|
104
|
+
* 1080p generation plus reference-to-video inputs. Source-video edit
|
|
105
|
+
* and extend are `grok-imagine-video` only.
|
|
100
106
|
*
|
|
101
107
|
* The Imagine video endpoints are not part of the OpenAI SDK surface (and
|
|
102
108
|
* xAI rejects the SDK's multipart paths), so requests are plain JSON calls
|
|
@@ -109,13 +115,21 @@ function buildGrokVideoUsage(
|
|
|
109
115
|
* - Aspect-ratio sizing via the "aspectRatio_resolution" size template
|
|
110
116
|
* (e.g. '16:9_720p'), consistent with the grok-imagine image models
|
|
111
117
|
* - Image-to-video via an `image` prompt part (starting frame URL or data URI)
|
|
118
|
+
* - Reference-to-video via image prompt parts with
|
|
119
|
+
* `metadata.role: 'reference'` or `'character'` (→ `reference_images`)
|
|
120
|
+
* and preset voices via `modelOptions.reference_audios`
|
|
121
|
+
* (grok-imagine-video-1.5 only)
|
|
122
|
+
* - Video editing / extension on `grok-imagine-video` via a source
|
|
123
|
+
* `video` prompt part and `modelOptions.mode: 'edit' | 'extend'`
|
|
124
|
+
* (`/v1/videos/edits` / `/v1/videos/extensions`; in extend mode
|
|
125
|
+
* `duration` is the added tail)
|
|
112
126
|
* - Usage reporting: billed seconds (`unitsBilled`) and exact cost
|
|
113
127
|
*/
|
|
114
128
|
export class GrokVideoAdapter<
|
|
115
129
|
TModel extends GrokVideoModel,
|
|
116
130
|
> extends BaseVideoAdapter<
|
|
117
131
|
TModel,
|
|
118
|
-
|
|
132
|
+
GrokVideoModelProviderOptionsByName[TModel],
|
|
119
133
|
GrokVideoModelProviderOptionsByName,
|
|
120
134
|
GrokVideoModelSizeByName,
|
|
121
135
|
GrokVideoModelInputModalitiesByName,
|
|
@@ -174,98 +188,350 @@ export class GrokVideoAdapter<
|
|
|
174
188
|
|
|
175
189
|
async createVideoJob(
|
|
176
190
|
options: VideoGenerationOptions<
|
|
177
|
-
|
|
191
|
+
GrokVideoModelProviderOptionsByName[TModel],
|
|
178
192
|
GrokVideoModelSizeByName[TModel],
|
|
179
193
|
GrokVideoModelDurationByName[TModel]
|
|
180
194
|
>,
|
|
181
195
|
): Promise<VideoJobResult> {
|
|
182
196
|
const { model, size, modelOptions, logger } = options
|
|
183
197
|
|
|
198
|
+
// `mode` is a routing hint for this adapter, not an API field — strip it
|
|
199
|
+
// before the remaining options are spread onto the request body. The
|
|
200
|
+
// per-model map narrows what callers can pass, but modelOptions often
|
|
201
|
+
// arrives as deserialized JSON, so the adapter handles the widest option
|
|
202
|
+
// surface (the 1.5 shape) uniformly and gates by model at runtime.
|
|
203
|
+
const { mode, ...wireOptions } = (modelOptions ??
|
|
204
|
+
{}) as GrokVideoRuntimeOptions
|
|
205
|
+
|
|
206
|
+
// `mode` is typed 'edit' | 'extend' but reaches us untrusted from JSON
|
|
207
|
+
// callers. An unrecognised value must not fall through to the
|
|
208
|
+
// generations endpoint with a source-video body — that would silently
|
|
209
|
+
// run (and bill) a generation the caller never asked for.
|
|
210
|
+
if (mode !== undefined && mode !== 'edit' && mode !== 'extend') {
|
|
211
|
+
throw new Error(
|
|
212
|
+
`${this.name}: unknown modelOptions.mode '${String(mode)}'. ` +
|
|
213
|
+
`Expected 'edit' or 'extend'.`,
|
|
214
|
+
)
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// The interleaved prompt decomposes into verbatim text plus typed media
|
|
218
|
+
// buckets. Reference audio is voice-id based (not an audio file), so
|
|
219
|
+
// audio prompt parts have no request field to land in.
|
|
220
|
+
const resolved = resolveMediaPrompt(options.prompt)
|
|
221
|
+
if (resolved.audios.length > 0) {
|
|
222
|
+
throw new Error(
|
|
223
|
+
`${this.name}.createVideoJob does not support audio prompt parts (model: ${model}). ` +
|
|
224
|
+
`To reference a preset voice, pass modelOptions.reference_audios ` +
|
|
225
|
+
`(e.g. [{ voice_id: 'eve' }]).`,
|
|
226
|
+
)
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
// A video prompt part is the source clip for edit / extension mode.
|
|
230
|
+
// Those endpoints are grok-imagine-video only — 1.5 has no video input.
|
|
231
|
+
if (
|
|
232
|
+
!isGrokVideoSourceModel(model) &&
|
|
233
|
+
(mode !== undefined || resolved.videos.length > 0)
|
|
234
|
+
) {
|
|
235
|
+
throw new Error(
|
|
236
|
+
`${this.name}: ${model} does not support video editing or extension. ` +
|
|
237
|
+
`Use 'grok-imagine-video' for /v1/videos/edits and /v1/videos/extensions.`,
|
|
238
|
+
)
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
// The mode must be chosen explicitly because the two endpoints have
|
|
242
|
+
// different semantics (edit rewrites the clip, extend appends
|
|
243
|
+
// `duration` seconds).
|
|
244
|
+
if (resolved.videos.length > 1) {
|
|
245
|
+
throw new Error(
|
|
246
|
+
`${this.name}: ${model} accepts at most one source video; received ${resolved.videos.length}.`,
|
|
247
|
+
)
|
|
248
|
+
}
|
|
249
|
+
const [sourceVideo] = resolved.videos
|
|
250
|
+
if (sourceVideo && mode === undefined) {
|
|
251
|
+
throw new Error(
|
|
252
|
+
`${this.name}: a video prompt part needs modelOptions.mode set to ` +
|
|
253
|
+
`'edit' (rewrite the clip) or 'extend' (append to it).`,
|
|
254
|
+
)
|
|
255
|
+
}
|
|
256
|
+
if (!sourceVideo && mode !== undefined) {
|
|
257
|
+
throw new Error(
|
|
258
|
+
`${this.name}: modelOptions.mode '${mode}' requires a video prompt ` +
|
|
259
|
+
`part carrying the source clip.`,
|
|
260
|
+
)
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
if (mode !== undefined && sourceVideo) {
|
|
264
|
+
return await this.createSourceVideoJob({
|
|
265
|
+
model,
|
|
266
|
+
mode,
|
|
267
|
+
sourceVideo,
|
|
268
|
+
resolved,
|
|
269
|
+
wireOptions,
|
|
270
|
+
size,
|
|
271
|
+
genericDuration: options.duration,
|
|
272
|
+
logger,
|
|
273
|
+
})
|
|
274
|
+
}
|
|
275
|
+
|
|
184
276
|
validateVideoSize(model, size)
|
|
185
277
|
|
|
278
|
+
// Pull the specially-handled keys out of the wire options: `duration`
|
|
279
|
+
// is folded into the snapped value below, and the reference fields are
|
|
280
|
+
// re-added explicitly so a JSON-serialized `null` or empty array reads
|
|
281
|
+
// as "unset" instead of leaking onto the wire.
|
|
282
|
+
const {
|
|
283
|
+
duration: rawOptionDuration,
|
|
284
|
+
reference_images: explicitReferenceImages,
|
|
285
|
+
reference_audios: referenceAudios,
|
|
286
|
+
...generationOptions
|
|
287
|
+
} = wireOptions
|
|
288
|
+
|
|
186
289
|
// Coerce the requested duration into the model's valid range (1–15s,
|
|
187
290
|
// integer) instead of rejecting it — `snapDuration` clamps and rounds.
|
|
188
291
|
// modelOptions wins over the generic `duration`, mirroring the size
|
|
189
292
|
// precedence below.
|
|
190
|
-
const rawDuration =
|
|
293
|
+
const rawDuration = rawOptionDuration ?? options.duration
|
|
191
294
|
const duration =
|
|
192
|
-
rawDuration
|
|
193
|
-
|
|
194
|
-
//
|
|
195
|
-
//
|
|
196
|
-
//
|
|
197
|
-
|
|
198
|
-
|
|
295
|
+
rawDuration != null ? this.snapDuration(rawDuration) : undefined
|
|
296
|
+
|
|
297
|
+
// Image parts split by role: un-roled / 'start_frame' images become the
|
|
298
|
+
// starting frame (image-to-video); 'reference' / 'character' images
|
|
299
|
+
// become reference_images (reference-to-video). The Imagine API has no
|
|
300
|
+
// mask / control / end-frame inputs. Unknown role strings (possible via
|
|
301
|
+
// JSON callers) throw rather than silently dropping the part.
|
|
302
|
+
const startFrames: Array<ImagePart<MediaInputMetadata>> = []
|
|
303
|
+
const referenceImages: Array<{ url: string }> = []
|
|
304
|
+
for (const part of resolved.images) {
|
|
305
|
+
const role = part.metadata?.role
|
|
306
|
+
switch (role) {
|
|
307
|
+
case 'mask':
|
|
308
|
+
case 'control':
|
|
309
|
+
case 'end_frame':
|
|
310
|
+
throw new Error(
|
|
311
|
+
`${this.name}: the Imagine video API has no '${role}' image ` +
|
|
312
|
+
`input on model ${model}. Use an un-roled / 'start_frame' ` +
|
|
313
|
+
`image as the starting frame, or 'reference' images.`,
|
|
314
|
+
)
|
|
315
|
+
case 'reference':
|
|
316
|
+
case 'character':
|
|
317
|
+
referenceImages.push({ url: mediaPartToUrl(part) })
|
|
318
|
+
break
|
|
319
|
+
case 'start_frame':
|
|
320
|
+
case undefined:
|
|
321
|
+
startFrames.push(part)
|
|
322
|
+
break
|
|
323
|
+
default:
|
|
324
|
+
throw new Error(
|
|
325
|
+
`${this.name}: unknown image metadata.role '${String(role)}'. ` +
|
|
326
|
+
`Expected 'start_frame', 'reference', or 'character'.`,
|
|
327
|
+
)
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
if (startFrames.length > 1) {
|
|
199
331
|
throw new Error(
|
|
200
|
-
`${this.name}
|
|
332
|
+
`${this.name}: ${model} accepts at most one starting-frame image; received ${startFrames.length}. ` +
|
|
333
|
+
`Use metadata.role: 'reference' for reference-to-video inputs.`,
|
|
201
334
|
)
|
|
202
335
|
}
|
|
203
|
-
|
|
336
|
+
// Explicit modelOptions.reference_images replaces the part-derived list
|
|
337
|
+
// (an explicit empty array means "none").
|
|
338
|
+
const finalReferenceImages =
|
|
339
|
+
explicitReferenceImages ??
|
|
340
|
+
(referenceImages.length > 0 ? referenceImages : undefined)
|
|
341
|
+
const referenceImageCount = finalReferenceImages?.length ?? 0
|
|
342
|
+
const referenceAudioCount = referenceAudios?.length ?? 0
|
|
343
|
+
const hasReference = referenceImageCount > 0 || referenceAudioCount > 0
|
|
344
|
+
|
|
345
|
+
// Reference inputs are a grok-imagine-video-1.5 feature. The per-model
|
|
346
|
+
// options map already hides the fields from other models at compile
|
|
347
|
+
// time; this runtime gate covers prompt-part roles and untyped callers.
|
|
348
|
+
if (!isGrokVideoReferenceModel(model) && hasReference) {
|
|
204
349
|
throw new Error(
|
|
205
|
-
`${this.name}
|
|
350
|
+
`${this.name}: ${model} does not support reference-to-video inputs. ` +
|
|
351
|
+
`Use 'grok-imagine-video-1.5' for reference_images / reference_audios.`,
|
|
206
352
|
)
|
|
207
353
|
}
|
|
208
|
-
|
|
209
|
-
// rejected by the API, so fail fast with a clear, actionable message
|
|
210
|
-
// pointing at the model that does support text-to-video.
|
|
211
|
-
if (resolved.images.length === 0 && isImageToVideoOnlyModel(model)) {
|
|
354
|
+
if (referenceAudioCount > GROK_VIDEO_MAX_REFERENCE_AUDIOS) {
|
|
212
355
|
throw new Error(
|
|
213
|
-
`${this.name}: ${model}
|
|
214
|
-
`Include an image prompt part as the starting frame, or use 'grok-imagine-video' for text-to-video.`,
|
|
356
|
+
`${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_AUDIOS} reference voices; received ${referenceAudioCount}.`,
|
|
215
357
|
)
|
|
216
358
|
}
|
|
217
|
-
if (
|
|
359
|
+
if (referenceImageCount > GROK_VIDEO_MAX_REFERENCE_IMAGES) {
|
|
218
360
|
throw new Error(
|
|
219
|
-
`${this.name}: ${model} accepts at most
|
|
361
|
+
`${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_IMAGES} reference images; received ${referenceImageCount}.`,
|
|
220
362
|
)
|
|
221
363
|
}
|
|
222
364
|
|
|
223
365
|
// Image-to-video: the single image prompt part becomes the starting frame
|
|
224
366
|
// and the prompt text describes the desired motion. URL sources are
|
|
225
367
|
// fetched by xAI's servers; data sources are sent as base64 data URIs.
|
|
226
|
-
const [startFrame] =
|
|
368
|
+
const [startFrame] = startFrames
|
|
369
|
+
|
|
370
|
+
// xAI rejects `image` + `reference_images` / `reference_audios` as a
|
|
371
|
+
// 400: only one of image-to-video or reference-to-video can be active.
|
|
372
|
+
if (startFrame && hasReference) {
|
|
373
|
+
throw new Error(
|
|
374
|
+
`${this.name}: image-to-video and reference-to-video cannot be combined. ` +
|
|
375
|
+
`Use a starting-frame image, or reference images / voices, not both.`,
|
|
376
|
+
)
|
|
377
|
+
}
|
|
227
378
|
|
|
228
379
|
// The generic `size` option carries an "aspectRatio_resolution" template
|
|
229
380
|
// (e.g. '16:9_720p') and maps to the Imagine API's `aspect_ratio` /
|
|
230
|
-
// `resolution` parameters; explicit modelOptions win over the template
|
|
381
|
+
// `resolution` parameters; explicit modelOptions win over the template
|
|
382
|
+
// (including `reference_images`, which replaces the part-derived list).
|
|
231
383
|
const parsedSize = size !== undefined ? parseGrokVideoSize(size) : undefined
|
|
384
|
+
const resolvedResolution =
|
|
385
|
+
generationOptions.resolution ?? parsedSize?.resolution
|
|
386
|
+
if (hasReference && resolvedResolution === '1080p') {
|
|
387
|
+
throw new Error(
|
|
388
|
+
`${this.name}: reference-to-video is capped at 720p on ${model}.`,
|
|
389
|
+
)
|
|
390
|
+
}
|
|
232
391
|
const request = {
|
|
233
392
|
model,
|
|
234
393
|
prompt: resolved.text,
|
|
235
|
-
...(startFrame && { image: { url:
|
|
394
|
+
...(startFrame && { image: { url: mediaPartToUrl(startFrame) } }),
|
|
395
|
+
...(referenceImageCount > 0 && {
|
|
396
|
+
reference_images: finalReferenceImages,
|
|
397
|
+
}),
|
|
398
|
+
...(referenceAudioCount > 0 && {
|
|
399
|
+
reference_audios: referenceAudios,
|
|
400
|
+
}),
|
|
236
401
|
...(parsedSize && {
|
|
237
402
|
aspect_ratio: parsedSize.aspectRatio,
|
|
238
403
|
...(parsedSize.resolution !== undefined && {
|
|
239
404
|
resolution: parsedSize.resolution,
|
|
240
405
|
}),
|
|
241
406
|
}),
|
|
242
|
-
|
|
243
|
-
//
|
|
244
|
-
//
|
|
407
|
+
// The remaining options spread after the size template so explicit
|
|
408
|
+
// aspect_ratio / resolution win over it; duration and the reference
|
|
409
|
+
// fields were destructured out above and re-added normalized.
|
|
410
|
+
...generationOptions,
|
|
245
411
|
...(duration !== undefined && { duration }),
|
|
246
412
|
}
|
|
247
413
|
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
414
|
+
return await this.postVideoJob('/videos/generations', request, {
|
|
415
|
+
model,
|
|
416
|
+
logger,
|
|
417
|
+
logLine: `activity=video.create provider=${this.name} model=${model} mode=generate size=${size ?? 'default'} duration=${duration ?? 'default'}`,
|
|
418
|
+
})
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
/**
|
|
422
|
+
* Build and post an edit / extension request. Both endpoints take only
|
|
423
|
+
* `model`, `prompt`, and the source `video` (plus `duration` — the length
|
|
424
|
+
* of the added tail — for extensions): output geometry is inherited from
|
|
425
|
+
* the source clip, capped at 720p, and edit outputs also inherit the
|
|
426
|
+
* source length. Rather than sending fields the API documents as ignored,
|
|
427
|
+
* the inapplicable options are rejected with actionable errors.
|
|
428
|
+
*/
|
|
429
|
+
private async createSourceVideoJob(args: {
|
|
430
|
+
model: string
|
|
431
|
+
mode: 'edit' | 'extend'
|
|
432
|
+
sourceVideo: VideoPart<MediaInputMetadata>
|
|
433
|
+
resolved: ReturnType<typeof resolveMediaPrompt>
|
|
434
|
+
wireOptions: Omit<GrokVideoRuntimeOptions, 'mode'>
|
|
435
|
+
size: string | undefined
|
|
436
|
+
genericDuration: number | undefined
|
|
437
|
+
logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']
|
|
438
|
+
}): Promise<VideoJobResult> {
|
|
439
|
+
const { model, mode, sourceVideo, resolved, wireOptions, logger } = args
|
|
440
|
+
const endpoint = mode === 'edit' ? '/videos/edits' : '/videos/extensions'
|
|
441
|
+
|
|
442
|
+
if (resolved.images.length > 0) {
|
|
443
|
+
throw new Error(
|
|
444
|
+
`${this.name}: '${mode}' mode takes only the source video — image ` +
|
|
445
|
+
`prompt parts are not supported by ${endpoint}.`,
|
|
252
446
|
)
|
|
447
|
+
}
|
|
253
448
|
|
|
254
|
-
|
|
449
|
+
// Pull every generation-only key out of the wire options so nothing can
|
|
450
|
+
// leak into the edit/extend body via the spread below. JSON-serialized
|
|
451
|
+
// `null` values (a common "unset" encoding) are treated as absent;
|
|
452
|
+
// actual values are rejected with actionable errors.
|
|
453
|
+
const {
|
|
454
|
+
aspect_ratio: aspectRatio,
|
|
455
|
+
resolution,
|
|
456
|
+
duration: modeDuration,
|
|
457
|
+
reference_images: referenceImagesOption,
|
|
458
|
+
reference_audios: referenceAudiosOption,
|
|
459
|
+
...passthrough
|
|
460
|
+
} = wireOptions
|
|
461
|
+
if (
|
|
462
|
+
(referenceImagesOption?.length ?? 0) > 0 ||
|
|
463
|
+
(referenceAudiosOption?.length ?? 0) > 0
|
|
464
|
+
) {
|
|
465
|
+
throw new Error(
|
|
466
|
+
`${this.name}: reference inputs are only supported by video ` +
|
|
467
|
+
`generation, not '${mode}' mode.`,
|
|
468
|
+
)
|
|
469
|
+
}
|
|
470
|
+
if (args.size !== undefined || aspectRatio != null || resolution != null) {
|
|
471
|
+
throw new Error(
|
|
472
|
+
`${this.name}: '${mode}' mode does not accept size / aspect_ratio / ` +
|
|
473
|
+
`resolution — the output inherits the source clip's geometry ` +
|
|
474
|
+
`(capped at 720p).`,
|
|
475
|
+
)
|
|
476
|
+
}
|
|
477
|
+
const rawDuration = modeDuration ?? args.genericDuration
|
|
478
|
+
if (mode === 'edit' && rawDuration != null) {
|
|
479
|
+
throw new Error(
|
|
480
|
+
`${this.name}: 'edit' mode does not accept a duration — the output ` +
|
|
481
|
+
`inherits the source clip's length. Use mode 'extend' to append ` +
|
|
482
|
+
`seconds to the clip.`,
|
|
483
|
+
)
|
|
484
|
+
}
|
|
485
|
+
// Extend: the snapped duration is the added-tail length (1–15s).
|
|
486
|
+
const duration =
|
|
487
|
+
rawDuration != null ? this.snapDuration(rawDuration) : undefined
|
|
488
|
+
|
|
489
|
+
const request = {
|
|
490
|
+
model,
|
|
491
|
+
prompt: resolved.text,
|
|
492
|
+
video: { url: mediaPartToUrl(sourceVideo) },
|
|
493
|
+
...passthrough,
|
|
494
|
+
...(duration !== undefined && { duration }),
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
return await this.postVideoJob(endpoint, request, {
|
|
498
|
+
model,
|
|
499
|
+
logger,
|
|
500
|
+
logLine: `activity=video.create provider=${this.name} model=${model} mode=${mode} duration=${duration ?? 'default'}`,
|
|
501
|
+
})
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
/**
|
|
505
|
+
* POST a create-job request body to one of the Imagine video endpoints
|
|
506
|
+
* (`/videos/generations`, `/videos/edits`, `/videos/extensions`) and read
|
|
507
|
+
* the `request_id` out of the shared response shape.
|
|
508
|
+
*/
|
|
509
|
+
private async postVideoJob(
|
|
510
|
+
endpoint: string,
|
|
511
|
+
request: Record<string, unknown>,
|
|
512
|
+
context: {
|
|
513
|
+
model: string
|
|
514
|
+
logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']
|
|
515
|
+
logLine: string
|
|
516
|
+
},
|
|
517
|
+
): Promise<VideoJobResult> {
|
|
518
|
+
const { model, logger, logLine } = context
|
|
519
|
+
try {
|
|
520
|
+
logger.request(logLine, { provider: this.name, model })
|
|
521
|
+
|
|
522
|
+
const response = await this.request(endpoint, {
|
|
255
523
|
method: 'POST',
|
|
256
524
|
body: JSON.stringify(request),
|
|
257
525
|
})
|
|
258
526
|
if (!response.ok) {
|
|
259
527
|
throw new Error(
|
|
260
|
-
`grok:
|
|
528
|
+
`grok: ${endpoint} request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,
|
|
261
529
|
)
|
|
262
530
|
}
|
|
263
531
|
|
|
264
532
|
const result = (await response.json()) as GrokVideoCreateResponse
|
|
265
533
|
if (!result.request_id) {
|
|
266
|
-
throw new Error(
|
|
267
|
-
'grok: video generation response contained no request_id',
|
|
268
|
-
)
|
|
534
|
+
throw new Error(`grok: ${endpoint} response contained no request_id`)
|
|
269
535
|
}
|
|
270
536
|
return { jobId: result.request_id, model }
|
|
271
537
|
} catch (error: unknown) {
|
|
@@ -394,15 +660,14 @@ export class GrokVideoAdapter<
|
|
|
394
660
|
*
|
|
395
661
|
* @experimental Video generation is an experimental feature and may change.
|
|
396
662
|
*
|
|
397
|
-
* @param model - The model name (e.g., 'grok-imagine-video')
|
|
663
|
+
* @param model - The model name (e.g., 'grok-imagine-video-1.5')
|
|
398
664
|
* @param apiKey - Your xAI API key
|
|
399
665
|
* @param config - Optional additional configuration
|
|
400
666
|
* @returns Configured Grok video adapter instance with resolved types
|
|
401
667
|
*
|
|
402
668
|
* @example
|
|
403
669
|
* ```typescript
|
|
404
|
-
*
|
|
405
|
-
* const adapter = createGrokVideo('grok-imagine-video', 'xai-...');
|
|
670
|
+
* const adapter = createGrokVideo('grok-imagine-video-1.5', 'xai-...');
|
|
406
671
|
*
|
|
407
672
|
* const { jobId } = await generateVideo({
|
|
408
673
|
* adapter,
|
|
@@ -440,7 +705,7 @@ export function createGrokVideo<TModel extends GrokVideoModel>(
|
|
|
440
705
|
* // Automatically uses XAI_API_KEY from environment
|
|
441
706
|
* const adapter = grokVideo('grok-imagine-video-1.5');
|
|
442
707
|
*
|
|
443
|
-
* // Image-to-video
|
|
708
|
+
* // Image-to-video: an optional image prompt part is the starting frame.
|
|
444
709
|
* const { jobId } = await generateVideo({
|
|
445
710
|
* adapter,
|
|
446
711
|
* prompt: [
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Grok Image Generation Provider Options
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* Provider-specific options for Grok image generation: the aspect-ratio
|
|
5
|
+
* sized Imagine API models (grok-imagine-image, grok-imagine-image-2.0,
|
|
6
|
+
* grok-imagine-image-quality) and the legacy pixel-sized grok-2-image-1212.
|
|
6
7
|
*/
|
|
7
8
|
|
|
8
9
|
/**
|
|
@@ -141,12 +142,25 @@ export interface GrokImagineImageProviderOptions extends GrokImageBaseProviderOp
|
|
|
141
142
|
service_tier?: 'default' | 'priority'
|
|
142
143
|
}
|
|
143
144
|
|
|
145
|
+
/**
|
|
146
|
+
* Provider options for grok-imagine-image-2.0, which adds a generation
|
|
147
|
+
* `quality` knob on top of the shared Imagine options.
|
|
148
|
+
*/
|
|
149
|
+
export interface GrokImagineImage2ProviderOptions extends GrokImagineImageProviderOptions {
|
|
150
|
+
/**
|
|
151
|
+
* Generation quality. Only supported by grok-imagine-image-2.0.
|
|
152
|
+
* @default 'medium'
|
|
153
|
+
*/
|
|
154
|
+
quality?: 'low' | 'medium'
|
|
155
|
+
}
|
|
156
|
+
|
|
144
157
|
/**
|
|
145
158
|
* Type-only map from model name to its specific provider options.
|
|
146
159
|
*/
|
|
147
160
|
export type GrokImageModelProviderOptionsByName = {
|
|
148
161
|
'grok-2-image-1212': GrokImageProviderOptions
|
|
149
162
|
'grok-imagine-image': GrokImagineImageProviderOptions
|
|
163
|
+
'grok-imagine-image-2.0': GrokImagineImage2ProviderOptions
|
|
150
164
|
'grok-imagine-image-quality': GrokImagineImageProviderOptions
|
|
151
165
|
}
|
|
152
166
|
|
|
@@ -156,6 +170,7 @@ export type GrokImageModelProviderOptionsByName = {
|
|
|
156
170
|
export type GrokImageModelSizeByName = {
|
|
157
171
|
'grok-2-image-1212': GrokImageSize
|
|
158
172
|
'grok-imagine-image': GrokImagineImageSize
|
|
173
|
+
'grok-imagine-image-2.0': GrokImagineImageSize
|
|
159
174
|
'grok-imagine-image-quality': GrokImagineImageSize
|
|
160
175
|
}
|
|
161
176
|
|
|
@@ -167,6 +182,7 @@ export type GrokImageModelSizeByName = {
|
|
|
167
182
|
export type GrokImageModelInputModalitiesByName = {
|
|
168
183
|
'grok-2-image-1212': readonly []
|
|
169
184
|
'grok-imagine-image': readonly ['image']
|
|
185
|
+
'grok-imagine-image-2.0': readonly ['image']
|
|
170
186
|
'grok-imagine-image-quality': readonly ['image']
|
|
171
187
|
}
|
|
172
188
|
|
package/src/index.ts
CHANGED
|
@@ -28,6 +28,8 @@ export {
|
|
|
28
28
|
} from './adapters/image'
|
|
29
29
|
export type {
|
|
30
30
|
GrokImageProviderOptions,
|
|
31
|
+
GrokImagineImageProviderOptions,
|
|
32
|
+
GrokImagineImage2ProviderOptions,
|
|
31
33
|
GrokImageModelProviderOptionsByName,
|
|
32
34
|
} from './image/image-provider-options'
|
|
33
35
|
|
|
@@ -43,7 +45,11 @@ export {
|
|
|
43
45
|
getGrokVideoDurationOptions,
|
|
44
46
|
} from './video/video-provider-options'
|
|
45
47
|
export type {
|
|
48
|
+
GrokVideoMode,
|
|
49
|
+
GrokVideoBaseProviderOptions,
|
|
50
|
+
GrokVideoSourceProviderOptions,
|
|
46
51
|
GrokVideoProviderOptions,
|
|
52
|
+
GrokVideoRuntimeOptions,
|
|
47
53
|
GrokVideoModelProviderOptionsByName,
|
|
48
54
|
GrokVideoModelSizeByName,
|
|
49
55
|
GrokVideoModelDurationByName,
|
|
@@ -101,6 +107,7 @@ export {
|
|
|
101
107
|
GROK_TTS_MODELS,
|
|
102
108
|
GROK_TRANSCRIPTION_MODELS,
|
|
103
109
|
GROK_REALTIME_MODELS,
|
|
110
|
+
GROK_DEFAULT_REALTIME_MODEL,
|
|
104
111
|
} from './model-meta'
|
|
105
112
|
export type {
|
|
106
113
|
GrokTextMetadata,
|