@tanstack/ai-grok 0.14.11 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.js +2 -2
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/transcription.js +4 -0
- package/dist/esm/adapters/transcription.js.map +1 -1
- package/dist/esm/adapters/tts.js +2 -1
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +34 -12
- package/dist/esm/adapters/video.js +134 -30
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/image/image-provider-options.d.ts +17 -2
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +3 -3
- package/dist/esm/index.js +2 -2
- package/dist/esm/model-meta.d.ts +49 -3
- package/dist/esm/model-meta.js +110 -8
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.js +17 -16
- package/dist/esm/realtime/adapter.js.map +1 -1
- package/dist/esm/realtime/token.d.ts +1 -1
- package/dist/esm/realtime/token.js +3 -2
- package/dist/esm/realtime/token.js.map +1 -1
- package/dist/esm/realtime/types.d.ts +1 -1
- package/dist/esm/tools/index.js +2 -0
- package/dist/esm/tools/index.js.map +1 -1
- package/dist/esm/video/video-provider-options.d.ts +131 -21
- package/dist/esm/video/video-provider-options.js +36 -10
- package/dist/esm/video/video-provider-options.js.map +1 -1
- package/package.json +6 -6
- package/src/adapters/image.ts +2 -1
- package/src/adapters/transcription.ts +2 -1
- package/src/adapters/video.ts +321 -53
- package/src/image/image-provider-options.ts +18 -2
- package/src/index.ts +7 -0
- package/src/model-meta.ts +109 -6
- package/src/realtime/adapter.ts +3 -2
- package/src/realtime/token.ts +4 -2
- package/src/realtime/types.ts +1 -1
- package/src/tools/index.ts +2 -0
- package/src/video/video-provider-options.ts +198 -34
package/src/adapters/video.ts
CHANGED
|
@@ -3,8 +3,11 @@ import { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'
|
|
|
3
3
|
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
4
4
|
import { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'
|
|
5
5
|
import {
|
|
6
|
+
GROK_VIDEO_MAX_REFERENCE_AUDIOS,
|
|
7
|
+
GROK_VIDEO_MAX_REFERENCE_IMAGES,
|
|
6
8
|
getGrokVideoDurationOptions,
|
|
7
|
-
|
|
9
|
+
isGrokVideoReferenceModel,
|
|
10
|
+
isGrokVideoSourceModel,
|
|
8
11
|
parseGrokVideoSize,
|
|
9
12
|
validateVideoSize,
|
|
10
13
|
} from '../video/video-provider-options'
|
|
@@ -15,6 +18,7 @@ import type {
|
|
|
15
18
|
TokenUsage,
|
|
16
19
|
VideoGenerationOptions,
|
|
17
20
|
VideoJobResult,
|
|
21
|
+
VideoPart,
|
|
18
22
|
VideoStatusResult,
|
|
19
23
|
VideoUrlResult,
|
|
20
24
|
} from '@tanstack/ai'
|
|
@@ -24,7 +28,7 @@ import type {
|
|
|
24
28
|
GrokVideoModelInputModalitiesByName,
|
|
25
29
|
GrokVideoModelProviderOptionsByName,
|
|
26
30
|
GrokVideoModelSizeByName,
|
|
27
|
-
|
|
31
|
+
GrokVideoRuntimeOptions,
|
|
28
32
|
} from '../video/video-provider-options'
|
|
29
33
|
import type { GrokClientConfig } from '../utils/client'
|
|
30
34
|
|
|
@@ -41,7 +45,7 @@ export interface GrokVideoConfig extends GrokClientConfig {}
|
|
|
41
45
|
*/
|
|
42
46
|
const USD_TICKS_PER_DOLLAR = 10_000_000_000
|
|
43
47
|
|
|
44
|
-
/** Response of POST /v1/videos/generations. */
|
|
48
|
+
/** Response of the POST /v1/videos/{generations,edits,extensions} endpoints. */
|
|
45
49
|
interface GrokVideoCreateResponse {
|
|
46
50
|
request_id?: string
|
|
47
51
|
}
|
|
@@ -62,11 +66,13 @@ interface GrokVideoStatusResponse {
|
|
|
62
66
|
}
|
|
63
67
|
|
|
64
68
|
/**
|
|
65
|
-
* Convert a TanStack
|
|
66
|
-
* video
|
|
67
|
-
* sources become base64 data URIs.
|
|
69
|
+
* Convert a TanStack image / video part to the URL string accepted by xAI's
|
|
70
|
+
* Imagine video endpoints: public URLs pass through (fetched by xAI's
|
|
71
|
+
* servers), data sources become base64 data URIs.
|
|
68
72
|
*/
|
|
69
|
-
function
|
|
73
|
+
function mediaPartToUrl(
|
|
74
|
+
part: ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata>,
|
|
75
|
+
): string {
|
|
70
76
|
if (part.source.type === 'url') return part.source.value
|
|
71
77
|
return `data:${part.source.mimeType};base64,${part.source.value}`
|
|
72
78
|
}
|
|
@@ -81,7 +87,10 @@ function buildGrokVideoUsage(
|
|
|
81
87
|
promptTokens: 0,
|
|
82
88
|
completionTokens: 0,
|
|
83
89
|
totalTokens: 0,
|
|
84
|
-
...(seconds !== undefined && {
|
|
90
|
+
...(seconds !== undefined && {
|
|
91
|
+
billed: { quantity: seconds, unit: 'seconds' },
|
|
92
|
+
unitsBilled: seconds,
|
|
93
|
+
}),
|
|
85
94
|
...(ticks !== undefined && { cost: ticks / USD_TICKS_PER_DOLLAR }),
|
|
86
95
|
}
|
|
87
96
|
}
|
|
@@ -93,10 +102,10 @@ function buildGrokVideoUsage(
|
|
|
93
102
|
* async jobs/polling architecture: create a generation request, poll it,
|
|
94
103
|
* then read the completed video URL.
|
|
95
104
|
*
|
|
96
|
-
*
|
|
97
|
-
* `grok-imagine-video-1.5` is
|
|
98
|
-
*
|
|
99
|
-
*
|
|
105
|
+
* Both models support text-to-video and image-to-video;
|
|
106
|
+
* `grok-imagine-video-1.5` is xAI's documented default and adds native
|
|
107
|
+
* 1080p generation plus reference-to-video inputs. Source-video edit
|
|
108
|
+
* and extend are `grok-imagine-video` only.
|
|
100
109
|
*
|
|
101
110
|
* The Imagine video endpoints are not part of the OpenAI SDK surface (and
|
|
102
111
|
* xAI rejects the SDK's multipart paths), so requests are plain JSON calls
|
|
@@ -109,13 +118,21 @@ function buildGrokVideoUsage(
|
|
|
109
118
|
* - Aspect-ratio sizing via the "aspectRatio_resolution" size template
|
|
110
119
|
* (e.g. '16:9_720p'), consistent with the grok-imagine image models
|
|
111
120
|
* - Image-to-video via an `image` prompt part (starting frame URL or data URI)
|
|
112
|
-
* -
|
|
121
|
+
* - Reference-to-video via image prompt parts with
|
|
122
|
+
* `metadata.role: 'reference'` or `'character'` (→ `reference_images`)
|
|
123
|
+
* and preset voices via `modelOptions.reference_audios`
|
|
124
|
+
* (grok-imagine-video-1.5 only)
|
|
125
|
+
* - Video editing / extension on `grok-imagine-video` via a source
|
|
126
|
+
* `video` prompt part and `modelOptions.mode: 'edit' | 'extend'`
|
|
127
|
+
* (`/v1/videos/edits` / `/v1/videos/extensions`; in extend mode
|
|
128
|
+
* `duration` is the added tail)
|
|
129
|
+
* - Usage reporting: billed seconds (`usage.billed`) and exact cost
|
|
113
130
|
*/
|
|
114
131
|
export class GrokVideoAdapter<
|
|
115
132
|
TModel extends GrokVideoModel,
|
|
116
133
|
> extends BaseVideoAdapter<
|
|
117
134
|
TModel,
|
|
118
|
-
|
|
135
|
+
GrokVideoModelProviderOptionsByName[TModel],
|
|
119
136
|
GrokVideoModelProviderOptionsByName,
|
|
120
137
|
GrokVideoModelSizeByName,
|
|
121
138
|
GrokVideoModelInputModalitiesByName,
|
|
@@ -174,98 +191,350 @@ export class GrokVideoAdapter<
|
|
|
174
191
|
|
|
175
192
|
async createVideoJob(
|
|
176
193
|
options: VideoGenerationOptions<
|
|
177
|
-
|
|
194
|
+
GrokVideoModelProviderOptionsByName[TModel],
|
|
178
195
|
GrokVideoModelSizeByName[TModel],
|
|
179
196
|
GrokVideoModelDurationByName[TModel]
|
|
180
197
|
>,
|
|
181
198
|
): Promise<VideoJobResult> {
|
|
182
199
|
const { model, size, modelOptions, logger } = options
|
|
183
200
|
|
|
201
|
+
// `mode` is a routing hint for this adapter, not an API field — strip it
|
|
202
|
+
// before the remaining options are spread onto the request body. The
|
|
203
|
+
// per-model map narrows what callers can pass, but modelOptions often
|
|
204
|
+
// arrives as deserialized JSON, so the adapter handles the widest option
|
|
205
|
+
// surface (the 1.5 shape) uniformly and gates by model at runtime.
|
|
206
|
+
const { mode, ...wireOptions } = (modelOptions ??
|
|
207
|
+
{}) as GrokVideoRuntimeOptions
|
|
208
|
+
|
|
209
|
+
// `mode` is typed 'edit' | 'extend' but reaches us untrusted from JSON
|
|
210
|
+
// callers. An unrecognised value must not fall through to the
|
|
211
|
+
// generations endpoint with a source-video body — that would silently
|
|
212
|
+
// run (and bill) a generation the caller never asked for.
|
|
213
|
+
if (mode !== undefined && mode !== 'edit' && mode !== 'extend') {
|
|
214
|
+
throw new Error(
|
|
215
|
+
`${this.name}: unknown modelOptions.mode '${String(mode)}'. ` +
|
|
216
|
+
`Expected 'edit' or 'extend'.`,
|
|
217
|
+
)
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// The interleaved prompt decomposes into verbatim text plus typed media
|
|
221
|
+
// buckets. Reference audio is voice-id based (not an audio file), so
|
|
222
|
+
// audio prompt parts have no request field to land in.
|
|
223
|
+
const resolved = resolveMediaPrompt(options.prompt)
|
|
224
|
+
if (resolved.audios.length > 0) {
|
|
225
|
+
throw new Error(
|
|
226
|
+
`${this.name}.createVideoJob does not support audio prompt parts (model: ${model}). ` +
|
|
227
|
+
`To reference a preset voice, pass modelOptions.reference_audios ` +
|
|
228
|
+
`(e.g. [{ voice_id: 'eve' }]).`,
|
|
229
|
+
)
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
// A video prompt part is the source clip for edit / extension mode.
|
|
233
|
+
// Those endpoints are grok-imagine-video only — 1.5 has no video input.
|
|
234
|
+
if (
|
|
235
|
+
!isGrokVideoSourceModel(model) &&
|
|
236
|
+
(mode !== undefined || resolved.videos.length > 0)
|
|
237
|
+
) {
|
|
238
|
+
throw new Error(
|
|
239
|
+
`${this.name}: ${model} does not support video editing or extension. ` +
|
|
240
|
+
`Use 'grok-imagine-video' for /v1/videos/edits and /v1/videos/extensions.`,
|
|
241
|
+
)
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
// The mode must be chosen explicitly because the two endpoints have
|
|
245
|
+
// different semantics (edit rewrites the clip, extend appends
|
|
246
|
+
// `duration` seconds).
|
|
247
|
+
if (resolved.videos.length > 1) {
|
|
248
|
+
throw new Error(
|
|
249
|
+
`${this.name}: ${model} accepts at most one source video; received ${resolved.videos.length}.`,
|
|
250
|
+
)
|
|
251
|
+
}
|
|
252
|
+
const [sourceVideo] = resolved.videos
|
|
253
|
+
if (sourceVideo && mode === undefined) {
|
|
254
|
+
throw new Error(
|
|
255
|
+
`${this.name}: a video prompt part needs modelOptions.mode set to ` +
|
|
256
|
+
`'edit' (rewrite the clip) or 'extend' (append to it).`,
|
|
257
|
+
)
|
|
258
|
+
}
|
|
259
|
+
if (!sourceVideo && mode !== undefined) {
|
|
260
|
+
throw new Error(
|
|
261
|
+
`${this.name}: modelOptions.mode '${mode}' requires a video prompt ` +
|
|
262
|
+
`part carrying the source clip.`,
|
|
263
|
+
)
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
if (mode !== undefined && sourceVideo) {
|
|
267
|
+
return await this.createSourceVideoJob({
|
|
268
|
+
model,
|
|
269
|
+
mode,
|
|
270
|
+
sourceVideo,
|
|
271
|
+
resolved,
|
|
272
|
+
wireOptions,
|
|
273
|
+
size,
|
|
274
|
+
genericDuration: options.duration,
|
|
275
|
+
logger,
|
|
276
|
+
})
|
|
277
|
+
}
|
|
278
|
+
|
|
184
279
|
validateVideoSize(model, size)
|
|
185
280
|
|
|
281
|
+
// Pull the specially-handled keys out of the wire options: `duration`
|
|
282
|
+
// is folded into the snapped value below, and the reference fields are
|
|
283
|
+
// re-added explicitly so a JSON-serialized `null` or empty array reads
|
|
284
|
+
// as "unset" instead of leaking onto the wire.
|
|
285
|
+
const {
|
|
286
|
+
duration: rawOptionDuration,
|
|
287
|
+
reference_images: explicitReferenceImages,
|
|
288
|
+
reference_audios: referenceAudios,
|
|
289
|
+
...generationOptions
|
|
290
|
+
} = wireOptions
|
|
291
|
+
|
|
186
292
|
// Coerce the requested duration into the model's valid range (1–15s,
|
|
187
293
|
// integer) instead of rejecting it — `snapDuration` clamps and rounds.
|
|
188
294
|
// modelOptions wins over the generic `duration`, mirroring the size
|
|
189
295
|
// precedence below.
|
|
190
|
-
const rawDuration =
|
|
296
|
+
const rawDuration = rawOptionDuration ?? options.duration
|
|
191
297
|
const duration =
|
|
192
|
-
rawDuration
|
|
193
|
-
|
|
194
|
-
//
|
|
195
|
-
//
|
|
196
|
-
//
|
|
197
|
-
|
|
198
|
-
|
|
298
|
+
rawDuration != null ? this.snapDuration(rawDuration) : undefined
|
|
299
|
+
|
|
300
|
+
// Image parts split by role: un-roled / 'start_frame' images become the
|
|
301
|
+
// starting frame (image-to-video); 'reference' / 'character' images
|
|
302
|
+
// become reference_images (reference-to-video). The Imagine API has no
|
|
303
|
+
// mask / control / end-frame inputs. Unknown role strings (possible via
|
|
304
|
+
// JSON callers) throw rather than silently dropping the part.
|
|
305
|
+
const startFrames: Array<ImagePart<MediaInputMetadata>> = []
|
|
306
|
+
const referenceImages: Array<{ url: string }> = []
|
|
307
|
+
for (const part of resolved.images) {
|
|
308
|
+
const role = part.metadata?.role
|
|
309
|
+
switch (role) {
|
|
310
|
+
case 'mask':
|
|
311
|
+
case 'control':
|
|
312
|
+
case 'end_frame':
|
|
313
|
+
throw new Error(
|
|
314
|
+
`${this.name}: the Imagine video API has no '${role}' image ` +
|
|
315
|
+
`input on model ${model}. Use an un-roled / 'start_frame' ` +
|
|
316
|
+
`image as the starting frame, or 'reference' images.`,
|
|
317
|
+
)
|
|
318
|
+
case 'reference':
|
|
319
|
+
case 'character':
|
|
320
|
+
referenceImages.push({ url: mediaPartToUrl(part) })
|
|
321
|
+
break
|
|
322
|
+
case 'start_frame':
|
|
323
|
+
case undefined:
|
|
324
|
+
startFrames.push(part)
|
|
325
|
+
break
|
|
326
|
+
default:
|
|
327
|
+
throw new Error(
|
|
328
|
+
`${this.name}: unknown image metadata.role '${String(role)}'. ` +
|
|
329
|
+
`Expected 'start_frame', 'reference', or 'character'.`,
|
|
330
|
+
)
|
|
331
|
+
}
|
|
332
|
+
}
|
|
333
|
+
if (startFrames.length > 1) {
|
|
199
334
|
throw new Error(
|
|
200
|
-
`${this.name}
|
|
335
|
+
`${this.name}: ${model} accepts at most one starting-frame image; received ${startFrames.length}. ` +
|
|
336
|
+
`Use metadata.role: 'reference' for reference-to-video inputs.`,
|
|
201
337
|
)
|
|
202
338
|
}
|
|
203
|
-
|
|
339
|
+
// Explicit modelOptions.reference_images replaces the part-derived list
|
|
340
|
+
// (an explicit empty array means "none").
|
|
341
|
+
const finalReferenceImages =
|
|
342
|
+
explicitReferenceImages ??
|
|
343
|
+
(referenceImages.length > 0 ? referenceImages : undefined)
|
|
344
|
+
const referenceImageCount = finalReferenceImages?.length ?? 0
|
|
345
|
+
const referenceAudioCount = referenceAudios?.length ?? 0
|
|
346
|
+
const hasReference = referenceImageCount > 0 || referenceAudioCount > 0
|
|
347
|
+
|
|
348
|
+
// Reference inputs are a grok-imagine-video-1.5 feature. The per-model
|
|
349
|
+
// options map already hides the fields from other models at compile
|
|
350
|
+
// time; this runtime gate covers prompt-part roles and untyped callers.
|
|
351
|
+
if (!isGrokVideoReferenceModel(model) && hasReference) {
|
|
204
352
|
throw new Error(
|
|
205
|
-
`${this.name}
|
|
353
|
+
`${this.name}: ${model} does not support reference-to-video inputs. ` +
|
|
354
|
+
`Use 'grok-imagine-video-1.5' for reference_images / reference_audios.`,
|
|
206
355
|
)
|
|
207
356
|
}
|
|
208
|
-
|
|
209
|
-
// rejected by the API, so fail fast with a clear, actionable message
|
|
210
|
-
// pointing at the model that does support text-to-video.
|
|
211
|
-
if (resolved.images.length === 0 && isImageToVideoOnlyModel(model)) {
|
|
357
|
+
if (referenceAudioCount > GROK_VIDEO_MAX_REFERENCE_AUDIOS) {
|
|
212
358
|
throw new Error(
|
|
213
|
-
`${this.name}: ${model}
|
|
214
|
-
`Include an image prompt part as the starting frame, or use 'grok-imagine-video' for text-to-video.`,
|
|
359
|
+
`${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_AUDIOS} reference voices; received ${referenceAudioCount}.`,
|
|
215
360
|
)
|
|
216
361
|
}
|
|
217
|
-
if (
|
|
362
|
+
if (referenceImageCount > GROK_VIDEO_MAX_REFERENCE_IMAGES) {
|
|
218
363
|
throw new Error(
|
|
219
|
-
`${this.name}: ${model} accepts at most
|
|
364
|
+
`${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_IMAGES} reference images; received ${referenceImageCount}.`,
|
|
220
365
|
)
|
|
221
366
|
}
|
|
222
367
|
|
|
223
368
|
// Image-to-video: the single image prompt part becomes the starting frame
|
|
224
369
|
// and the prompt text describes the desired motion. URL sources are
|
|
225
370
|
// fetched by xAI's servers; data sources are sent as base64 data URIs.
|
|
226
|
-
const [startFrame] =
|
|
371
|
+
const [startFrame] = startFrames
|
|
372
|
+
|
|
373
|
+
// xAI rejects `image` + `reference_images` / `reference_audios` as a
|
|
374
|
+
// 400: only one of image-to-video or reference-to-video can be active.
|
|
375
|
+
if (startFrame && hasReference) {
|
|
376
|
+
throw new Error(
|
|
377
|
+
`${this.name}: image-to-video and reference-to-video cannot be combined. ` +
|
|
378
|
+
`Use a starting-frame image, or reference images / voices, not both.`,
|
|
379
|
+
)
|
|
380
|
+
}
|
|
227
381
|
|
|
228
382
|
// The generic `size` option carries an "aspectRatio_resolution" template
|
|
229
383
|
// (e.g. '16:9_720p') and maps to the Imagine API's `aspect_ratio` /
|
|
230
|
-
// `resolution` parameters; explicit modelOptions win over the template
|
|
384
|
+
// `resolution` parameters; explicit modelOptions win over the template
|
|
385
|
+
// (including `reference_images`, which replaces the part-derived list).
|
|
231
386
|
const parsedSize = size !== undefined ? parseGrokVideoSize(size) : undefined
|
|
387
|
+
const resolvedResolution =
|
|
388
|
+
generationOptions.resolution ?? parsedSize?.resolution
|
|
389
|
+
if (hasReference && resolvedResolution === '1080p') {
|
|
390
|
+
throw new Error(
|
|
391
|
+
`${this.name}: reference-to-video is capped at 720p on ${model}.`,
|
|
392
|
+
)
|
|
393
|
+
}
|
|
232
394
|
const request = {
|
|
233
395
|
model,
|
|
234
396
|
prompt: resolved.text,
|
|
235
|
-
...(startFrame && { image: { url:
|
|
397
|
+
...(startFrame && { image: { url: mediaPartToUrl(startFrame) } }),
|
|
398
|
+
...(referenceImageCount > 0 && {
|
|
399
|
+
reference_images: finalReferenceImages,
|
|
400
|
+
}),
|
|
401
|
+
...(referenceAudioCount > 0 && {
|
|
402
|
+
reference_audios: referenceAudios,
|
|
403
|
+
}),
|
|
236
404
|
...(parsedSize && {
|
|
237
405
|
aspect_ratio: parsedSize.aspectRatio,
|
|
238
406
|
...(parsedSize.resolution !== undefined && {
|
|
239
407
|
resolution: parsedSize.resolution,
|
|
240
408
|
}),
|
|
241
409
|
}),
|
|
242
|
-
|
|
243
|
-
//
|
|
244
|
-
//
|
|
410
|
+
// The remaining options spread after the size template so explicit
|
|
411
|
+
// aspect_ratio / resolution win over it; duration and the reference
|
|
412
|
+
// fields were destructured out above and re-added normalized.
|
|
413
|
+
...generationOptions,
|
|
245
414
|
...(duration !== undefined && { duration }),
|
|
246
415
|
}
|
|
247
416
|
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
417
|
+
return await this.postVideoJob('/videos/generations', request, {
|
|
418
|
+
model,
|
|
419
|
+
logger,
|
|
420
|
+
logLine: `activity=video.create provider=${this.name} model=${model} mode=generate size=${size ?? 'default'} duration=${duration ?? 'default'}`,
|
|
421
|
+
})
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
/**
|
|
425
|
+
* Build and post an edit / extension request. Both endpoints take only
|
|
426
|
+
* `model`, `prompt`, and the source `video` (plus `duration` — the length
|
|
427
|
+
* of the added tail — for extensions): output geometry is inherited from
|
|
428
|
+
* the source clip, capped at 720p, and edit outputs also inherit the
|
|
429
|
+
* source length. Rather than sending fields the API documents as ignored,
|
|
430
|
+
* the inapplicable options are rejected with actionable errors.
|
|
431
|
+
*/
|
|
432
|
+
private async createSourceVideoJob(args: {
|
|
433
|
+
model: string
|
|
434
|
+
mode: 'edit' | 'extend'
|
|
435
|
+
sourceVideo: VideoPart<MediaInputMetadata>
|
|
436
|
+
resolved: ReturnType<typeof resolveMediaPrompt>
|
|
437
|
+
wireOptions: Omit<GrokVideoRuntimeOptions, 'mode'>
|
|
438
|
+
size: string | undefined
|
|
439
|
+
genericDuration: number | undefined
|
|
440
|
+
logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']
|
|
441
|
+
}): Promise<VideoJobResult> {
|
|
442
|
+
const { model, mode, sourceVideo, resolved, wireOptions, logger } = args
|
|
443
|
+
const endpoint = mode === 'edit' ? '/videos/edits' : '/videos/extensions'
|
|
444
|
+
|
|
445
|
+
if (resolved.images.length > 0) {
|
|
446
|
+
throw new Error(
|
|
447
|
+
`${this.name}: '${mode}' mode takes only the source video — image ` +
|
|
448
|
+
`prompt parts are not supported by ${endpoint}.`,
|
|
252
449
|
)
|
|
450
|
+
}
|
|
253
451
|
|
|
254
|
-
|
|
452
|
+
// Pull every generation-only key out of the wire options so nothing can
|
|
453
|
+
// leak into the edit/extend body via the spread below. JSON-serialized
|
|
454
|
+
// `null` values (a common "unset" encoding) are treated as absent;
|
|
455
|
+
// actual values are rejected with actionable errors.
|
|
456
|
+
const {
|
|
457
|
+
aspect_ratio: aspectRatio,
|
|
458
|
+
resolution,
|
|
459
|
+
duration: modeDuration,
|
|
460
|
+
reference_images: referenceImagesOption,
|
|
461
|
+
reference_audios: referenceAudiosOption,
|
|
462
|
+
...passthrough
|
|
463
|
+
} = wireOptions
|
|
464
|
+
if (
|
|
465
|
+
(referenceImagesOption?.length ?? 0) > 0 ||
|
|
466
|
+
(referenceAudiosOption?.length ?? 0) > 0
|
|
467
|
+
) {
|
|
468
|
+
throw new Error(
|
|
469
|
+
`${this.name}: reference inputs are only supported by video ` +
|
|
470
|
+
`generation, not '${mode}' mode.`,
|
|
471
|
+
)
|
|
472
|
+
}
|
|
473
|
+
if (args.size !== undefined || aspectRatio != null || resolution != null) {
|
|
474
|
+
throw new Error(
|
|
475
|
+
`${this.name}: '${mode}' mode does not accept size / aspect_ratio / ` +
|
|
476
|
+
`resolution — the output inherits the source clip's geometry ` +
|
|
477
|
+
`(capped at 720p).`,
|
|
478
|
+
)
|
|
479
|
+
}
|
|
480
|
+
const rawDuration = modeDuration ?? args.genericDuration
|
|
481
|
+
if (mode === 'edit' && rawDuration != null) {
|
|
482
|
+
throw new Error(
|
|
483
|
+
`${this.name}: 'edit' mode does not accept a duration — the output ` +
|
|
484
|
+
`inherits the source clip's length. Use mode 'extend' to append ` +
|
|
485
|
+
`seconds to the clip.`,
|
|
486
|
+
)
|
|
487
|
+
}
|
|
488
|
+
// Extend: the snapped duration is the added-tail length (1–15s).
|
|
489
|
+
const duration =
|
|
490
|
+
rawDuration != null ? this.snapDuration(rawDuration) : undefined
|
|
491
|
+
|
|
492
|
+
const request = {
|
|
493
|
+
model,
|
|
494
|
+
prompt: resolved.text,
|
|
495
|
+
video: { url: mediaPartToUrl(sourceVideo) },
|
|
496
|
+
...passthrough,
|
|
497
|
+
...(duration !== undefined && { duration }),
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
return await this.postVideoJob(endpoint, request, {
|
|
501
|
+
model,
|
|
502
|
+
logger,
|
|
503
|
+
logLine: `activity=video.create provider=${this.name} model=${model} mode=${mode} duration=${duration ?? 'default'}`,
|
|
504
|
+
})
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
/**
|
|
508
|
+
* POST a create-job request body to one of the Imagine video endpoints
|
|
509
|
+
* (`/videos/generations`, `/videos/edits`, `/videos/extensions`) and read
|
|
510
|
+
* the `request_id` out of the shared response shape.
|
|
511
|
+
*/
|
|
512
|
+
private async postVideoJob(
|
|
513
|
+
endpoint: string,
|
|
514
|
+
request: Record<string, unknown>,
|
|
515
|
+
context: {
|
|
516
|
+
model: string
|
|
517
|
+
logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']
|
|
518
|
+
logLine: string
|
|
519
|
+
},
|
|
520
|
+
): Promise<VideoJobResult> {
|
|
521
|
+
const { model, logger, logLine } = context
|
|
522
|
+
try {
|
|
523
|
+
logger.request(logLine, { provider: this.name, model })
|
|
524
|
+
|
|
525
|
+
const response = await this.request(endpoint, {
|
|
255
526
|
method: 'POST',
|
|
256
527
|
body: JSON.stringify(request),
|
|
257
528
|
})
|
|
258
529
|
if (!response.ok) {
|
|
259
530
|
throw new Error(
|
|
260
|
-
`grok:
|
|
531
|
+
`grok: ${endpoint} request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,
|
|
261
532
|
)
|
|
262
533
|
}
|
|
263
534
|
|
|
264
535
|
const result = (await response.json()) as GrokVideoCreateResponse
|
|
265
536
|
if (!result.request_id) {
|
|
266
|
-
throw new Error(
|
|
267
|
-
'grok: video generation response contained no request_id',
|
|
268
|
-
)
|
|
537
|
+
throw new Error(`grok: ${endpoint} response contained no request_id`)
|
|
269
538
|
}
|
|
270
539
|
return { jobId: result.request_id, model }
|
|
271
540
|
} catch (error: unknown) {
|
|
@@ -394,15 +663,14 @@ export class GrokVideoAdapter<
|
|
|
394
663
|
*
|
|
395
664
|
* @experimental Video generation is an experimental feature and may change.
|
|
396
665
|
*
|
|
397
|
-
* @param model - The model name (e.g., 'grok-imagine-video')
|
|
666
|
+
* @param model - The model name (e.g., 'grok-imagine-video-1.5')
|
|
398
667
|
* @param apiKey - Your xAI API key
|
|
399
668
|
* @param config - Optional additional configuration
|
|
400
669
|
* @returns Configured Grok video adapter instance with resolved types
|
|
401
670
|
*
|
|
402
671
|
* @example
|
|
403
672
|
* ```typescript
|
|
404
|
-
*
|
|
405
|
-
* const adapter = createGrokVideo('grok-imagine-video', 'xai-...');
|
|
673
|
+
* const adapter = createGrokVideo('grok-imagine-video-1.5', 'xai-...');
|
|
406
674
|
*
|
|
407
675
|
* const { jobId } = await generateVideo({
|
|
408
676
|
* adapter,
|
|
@@ -440,7 +708,7 @@ export function createGrokVideo<TModel extends GrokVideoModel>(
|
|
|
440
708
|
* // Automatically uses XAI_API_KEY from environment
|
|
441
709
|
* const adapter = grokVideo('grok-imagine-video-1.5');
|
|
442
710
|
*
|
|
443
|
-
* // Image-to-video
|
|
711
|
+
* // Image-to-video: an optional image prompt part is the starting frame.
|
|
444
712
|
* const { jobId } = await generateVideo({
|
|
445
713
|
* adapter,
|
|
446
714
|
* prompt: [
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Grok Image Generation Provider Options
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* Provider-specific options for Grok image generation: the aspect-ratio
|
|
5
|
+
* sized Imagine API models (grok-imagine-image, grok-imagine-image-2.0,
|
|
6
|
+
* grok-imagine-image-quality) and the legacy pixel-sized grok-2-image-1212.
|
|
6
7
|
*/
|
|
7
8
|
|
|
8
9
|
/**
|
|
@@ -141,12 +142,25 @@ export interface GrokImagineImageProviderOptions extends GrokImageBaseProviderOp
|
|
|
141
142
|
service_tier?: 'default' | 'priority'
|
|
142
143
|
}
|
|
143
144
|
|
|
145
|
+
/**
|
|
146
|
+
* Provider options for grok-imagine-image-2.0, which adds a generation
|
|
147
|
+
* `quality` knob on top of the shared Imagine options.
|
|
148
|
+
*/
|
|
149
|
+
export interface GrokImagineImage2ProviderOptions extends GrokImagineImageProviderOptions {
|
|
150
|
+
/**
|
|
151
|
+
* Generation quality. Only supported by grok-imagine-image-2.0.
|
|
152
|
+
* @default 'medium'
|
|
153
|
+
*/
|
|
154
|
+
quality?: 'low' | 'medium'
|
|
155
|
+
}
|
|
156
|
+
|
|
144
157
|
/**
|
|
145
158
|
* Type-only map from model name to its specific provider options.
|
|
146
159
|
*/
|
|
147
160
|
export type GrokImageModelProviderOptionsByName = {
|
|
148
161
|
'grok-2-image-1212': GrokImageProviderOptions
|
|
149
162
|
'grok-imagine-image': GrokImagineImageProviderOptions
|
|
163
|
+
'grok-imagine-image-2.0': GrokImagineImage2ProviderOptions
|
|
150
164
|
'grok-imagine-image-quality': GrokImagineImageProviderOptions
|
|
151
165
|
}
|
|
152
166
|
|
|
@@ -156,6 +170,7 @@ export type GrokImageModelProviderOptionsByName = {
|
|
|
156
170
|
export type GrokImageModelSizeByName = {
|
|
157
171
|
'grok-2-image-1212': GrokImageSize
|
|
158
172
|
'grok-imagine-image': GrokImagineImageSize
|
|
173
|
+
'grok-imagine-image-2.0': GrokImagineImageSize
|
|
159
174
|
'grok-imagine-image-quality': GrokImagineImageSize
|
|
160
175
|
}
|
|
161
176
|
|
|
@@ -167,6 +182,7 @@ export type GrokImageModelSizeByName = {
|
|
|
167
182
|
export type GrokImageModelInputModalitiesByName = {
|
|
168
183
|
'grok-2-image-1212': readonly []
|
|
169
184
|
'grok-imagine-image': readonly ['image']
|
|
185
|
+
'grok-imagine-image-2.0': readonly ['image']
|
|
170
186
|
'grok-imagine-image-quality': readonly ['image']
|
|
171
187
|
}
|
|
172
188
|
|
package/src/index.ts
CHANGED
|
@@ -28,6 +28,8 @@ export {
|
|
|
28
28
|
} from './adapters/image'
|
|
29
29
|
export type {
|
|
30
30
|
GrokImageProviderOptions,
|
|
31
|
+
GrokImagineImageProviderOptions,
|
|
32
|
+
GrokImagineImage2ProviderOptions,
|
|
31
33
|
GrokImageModelProviderOptionsByName,
|
|
32
34
|
} from './image/image-provider-options'
|
|
33
35
|
|
|
@@ -43,7 +45,11 @@ export {
|
|
|
43
45
|
getGrokVideoDurationOptions,
|
|
44
46
|
} from './video/video-provider-options'
|
|
45
47
|
export type {
|
|
48
|
+
GrokVideoMode,
|
|
49
|
+
GrokVideoBaseProviderOptions,
|
|
50
|
+
GrokVideoSourceProviderOptions,
|
|
46
51
|
GrokVideoProviderOptions,
|
|
52
|
+
GrokVideoRuntimeOptions,
|
|
47
53
|
GrokVideoModelProviderOptionsByName,
|
|
48
54
|
GrokVideoModelSizeByName,
|
|
49
55
|
GrokVideoModelDurationByName,
|
|
@@ -101,6 +107,7 @@ export {
|
|
|
101
107
|
GROK_TTS_MODELS,
|
|
102
108
|
GROK_TRANSCRIPTION_MODELS,
|
|
103
109
|
GROK_REALTIME_MODELS,
|
|
110
|
+
GROK_DEFAULT_REALTIME_MODEL,
|
|
104
111
|
} from './model-meta'
|
|
105
112
|
export type {
|
|
106
113
|
GrokTextMetadata,
|