@tanstack/ai-grok 0.14.11 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/esm/adapters/image.js +2 -2
  2. package/dist/esm/adapters/image.js.map +1 -1
  3. package/dist/esm/adapters/transcription.js +4 -0
  4. package/dist/esm/adapters/transcription.js.map +1 -1
  5. package/dist/esm/adapters/tts.js +2 -1
  6. package/dist/esm/adapters/tts.js.map +1 -1
  7. package/dist/esm/adapters/video.d.ts +34 -12
  8. package/dist/esm/adapters/video.js +134 -30
  9. package/dist/esm/adapters/video.js.map +1 -1
  10. package/dist/esm/image/image-provider-options.d.ts +17 -2
  11. package/dist/esm/image/image-provider-options.js.map +1 -1
  12. package/dist/esm/index.d.ts +3 -3
  13. package/dist/esm/index.js +2 -2
  14. package/dist/esm/model-meta.d.ts +49 -3
  15. package/dist/esm/model-meta.js +110 -8
  16. package/dist/esm/model-meta.js.map +1 -1
  17. package/dist/esm/realtime/adapter.js +17 -16
  18. package/dist/esm/realtime/adapter.js.map +1 -1
  19. package/dist/esm/realtime/token.d.ts +1 -1
  20. package/dist/esm/realtime/token.js +3 -2
  21. package/dist/esm/realtime/token.js.map +1 -1
  22. package/dist/esm/realtime/types.d.ts +1 -1
  23. package/dist/esm/tools/index.js +2 -0
  24. package/dist/esm/tools/index.js.map +1 -1
  25. package/dist/esm/video/video-provider-options.d.ts +131 -21
  26. package/dist/esm/video/video-provider-options.js +36 -10
  27. package/dist/esm/video/video-provider-options.js.map +1 -1
  28. package/package.json +6 -6
  29. package/src/adapters/image.ts +2 -1
  30. package/src/adapters/transcription.ts +2 -1
  31. package/src/adapters/video.ts +321 -53
  32. package/src/image/image-provider-options.ts +18 -2
  33. package/src/index.ts +7 -0
  34. package/src/model-meta.ts +109 -6
  35. package/src/realtime/adapter.ts +3 -2
  36. package/src/realtime/token.ts +4 -2
  37. package/src/realtime/types.ts +1 -1
  38. package/src/tools/index.ts +2 -0
  39. package/src/video/video-provider-options.ts +198 -34
@@ -3,8 +3,11 @@ import { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'
3
3
  import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
4
4
  import { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'
5
5
  import {
6
+ GROK_VIDEO_MAX_REFERENCE_AUDIOS,
7
+ GROK_VIDEO_MAX_REFERENCE_IMAGES,
6
8
  getGrokVideoDurationOptions,
7
- isImageToVideoOnlyModel,
9
+ isGrokVideoReferenceModel,
10
+ isGrokVideoSourceModel,
8
11
  parseGrokVideoSize,
9
12
  validateVideoSize,
10
13
  } from '../video/video-provider-options'
@@ -15,6 +18,7 @@ import type {
15
18
  TokenUsage,
16
19
  VideoGenerationOptions,
17
20
  VideoJobResult,
21
+ VideoPart,
18
22
  VideoStatusResult,
19
23
  VideoUrlResult,
20
24
  } from '@tanstack/ai'
@@ -24,7 +28,7 @@ import type {
24
28
  GrokVideoModelInputModalitiesByName,
25
29
  GrokVideoModelProviderOptionsByName,
26
30
  GrokVideoModelSizeByName,
27
- GrokVideoProviderOptions,
31
+ GrokVideoRuntimeOptions,
28
32
  } from '../video/video-provider-options'
29
33
  import type { GrokClientConfig } from '../utils/client'
30
34
 
@@ -41,7 +45,7 @@ export interface GrokVideoConfig extends GrokClientConfig {}
41
45
  */
42
46
  const USD_TICKS_PER_DOLLAR = 10_000_000_000
43
47
 
44
- /** Response of POST /v1/videos/generations. */
48
+ /** Response of the POST /v1/videos/{generations,edits,extensions} endpoints. */
45
49
  interface GrokVideoCreateResponse {
46
50
  request_id?: string
47
51
  }
@@ -62,11 +66,13 @@ interface GrokVideoStatusResponse {
62
66
  }
63
67
 
64
68
  /**
65
- * Convert a TanStack ImagePart to the URL string accepted by xAI's Imagine
66
- * video endpoint: public URLs pass through (fetched by xAI's servers), data
67
- * sources become base64 data URIs.
69
+ * Convert a TanStack image / video part to the URL string accepted by xAI's
70
+ * Imagine video endpoints: public URLs pass through (fetched by xAI's
71
+ * servers), data sources become base64 data URIs.
68
72
  */
69
- function imagePartToUrl(part: ImagePart<MediaInputMetadata>): string {
73
+ function mediaPartToUrl(
74
+ part: ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata>,
75
+ ): string {
70
76
  if (part.source.type === 'url') return part.source.value
71
77
  return `data:${part.source.mimeType};base64,${part.source.value}`
72
78
  }
@@ -81,7 +87,10 @@ function buildGrokVideoUsage(
81
87
  promptTokens: 0,
82
88
  completionTokens: 0,
83
89
  totalTokens: 0,
84
- ...(seconds !== undefined && { unitsBilled: seconds }),
90
+ ...(seconds !== undefined && {
91
+ billed: { quantity: seconds, unit: 'seconds' },
92
+ unitsBilled: seconds,
93
+ }),
85
94
  ...(ticks !== undefined && { cost: ticks / USD_TICKS_PER_DOLLAR }),
86
95
  }
87
96
  }
@@ -93,10 +102,10 @@ function buildGrokVideoUsage(
93
102
  * async jobs/polling architecture: create a generation request, poll it,
94
103
  * then read the completed video URL.
95
104
  *
96
- * `grok-imagine-video` (v1.0) supports text-to-video and image-to-video.
97
- * `grok-imagine-video-1.5` is image-to-video only — every request needs an
98
- * image prompt part as the starting frame, and the adapter rejects a
99
- * text-only prompt with a clear error rather than a raw API 400.
105
+ * Both models support text-to-video and image-to-video;
106
+ * `grok-imagine-video-1.5` is xAI's documented default and adds native
107
+ * 1080p generation plus reference-to-video inputs. Source-video edit
108
+ * and extend are `grok-imagine-video` only.
100
109
  *
101
110
  * The Imagine video endpoints are not part of the OpenAI SDK surface (and
102
111
  * xAI rejects the SDK's multipart paths), so requests are plain JSON calls
@@ -109,13 +118,21 @@ function buildGrokVideoUsage(
109
118
  * - Aspect-ratio sizing via the "aspectRatio_resolution" size template
110
119
  * (e.g. '16:9_720p'), consistent with the grok-imagine image models
111
120
  * - Image-to-video via an `image` prompt part (starting frame URL or data URI)
112
- * - Usage reporting: billed seconds (`unitsBilled`) and exact cost
121
+ * - Reference-to-video via image prompt parts with
122
+ * `metadata.role: 'reference'` or `'character'` (→ `reference_images`)
123
+ * and preset voices via `modelOptions.reference_audios`
124
+ * (grok-imagine-video-1.5 only)
125
+ * - Video editing / extension on `grok-imagine-video` via a source
126
+ * `video` prompt part and `modelOptions.mode: 'edit' | 'extend'`
127
+ * (`/v1/videos/edits` / `/v1/videos/extensions`; in extend mode
128
+ * `duration` is the added tail)
129
+ * - Usage reporting: billed seconds (`usage.billed`) and exact cost
113
130
  */
114
131
  export class GrokVideoAdapter<
115
132
  TModel extends GrokVideoModel,
116
133
  > extends BaseVideoAdapter<
117
134
  TModel,
118
- GrokVideoProviderOptions,
135
+ GrokVideoModelProviderOptionsByName[TModel],
119
136
  GrokVideoModelProviderOptionsByName,
120
137
  GrokVideoModelSizeByName,
121
138
  GrokVideoModelInputModalitiesByName,
@@ -174,98 +191,350 @@ export class GrokVideoAdapter<
174
191
 
175
192
  async createVideoJob(
176
193
  options: VideoGenerationOptions<
177
- GrokVideoProviderOptions,
194
+ GrokVideoModelProviderOptionsByName[TModel],
178
195
  GrokVideoModelSizeByName[TModel],
179
196
  GrokVideoModelDurationByName[TModel]
180
197
  >,
181
198
  ): Promise<VideoJobResult> {
182
199
  const { model, size, modelOptions, logger } = options
183
200
 
201
+ // `mode` is a routing hint for this adapter, not an API field — strip it
202
+ // before the remaining options are spread onto the request body. The
203
+ // per-model map narrows what callers can pass, but modelOptions often
204
+ // arrives as deserialized JSON, so the adapter handles the widest option
205
+ // surface (the 1.5 shape) uniformly and gates by model at runtime.
206
+ const { mode, ...wireOptions } = (modelOptions ??
207
+ {}) as GrokVideoRuntimeOptions
208
+
209
+ // `mode` is typed 'edit' | 'extend' but reaches us untrusted from JSON
210
+ // callers. An unrecognised value must not fall through to the
211
+ // generations endpoint with a source-video body — that would silently
212
+ // run (and bill) a generation the caller never asked for.
213
+ if (mode !== undefined && mode !== 'edit' && mode !== 'extend') {
214
+ throw new Error(
215
+ `${this.name}: unknown modelOptions.mode '${String(mode)}'. ` +
216
+ `Expected 'edit' or 'extend'.`,
217
+ )
218
+ }
219
+
220
+ // The interleaved prompt decomposes into verbatim text plus typed media
221
+ // buckets. Reference audio is voice-id based (not an audio file), so
222
+ // audio prompt parts have no request field to land in.
223
+ const resolved = resolveMediaPrompt(options.prompt)
224
+ if (resolved.audios.length > 0) {
225
+ throw new Error(
226
+ `${this.name}.createVideoJob does not support audio prompt parts (model: ${model}). ` +
227
+ `To reference a preset voice, pass modelOptions.reference_audios ` +
228
+ `(e.g. [{ voice_id: 'eve' }]).`,
229
+ )
230
+ }
231
+
232
+ // A video prompt part is the source clip for edit / extension mode.
233
+ // Those endpoints are grok-imagine-video only — 1.5 has no video input.
234
+ if (
235
+ !isGrokVideoSourceModel(model) &&
236
+ (mode !== undefined || resolved.videos.length > 0)
237
+ ) {
238
+ throw new Error(
239
+ `${this.name}: ${model} does not support video editing or extension. ` +
240
+ `Use 'grok-imagine-video' for /v1/videos/edits and /v1/videos/extensions.`,
241
+ )
242
+ }
243
+
244
+ // The mode must be chosen explicitly because the two endpoints have
245
+ // different semantics (edit rewrites the clip, extend appends
246
+ // `duration` seconds).
247
+ if (resolved.videos.length > 1) {
248
+ throw new Error(
249
+ `${this.name}: ${model} accepts at most one source video; received ${resolved.videos.length}.`,
250
+ )
251
+ }
252
+ const [sourceVideo] = resolved.videos
253
+ if (sourceVideo && mode === undefined) {
254
+ throw new Error(
255
+ `${this.name}: a video prompt part needs modelOptions.mode set to ` +
256
+ `'edit' (rewrite the clip) or 'extend' (append to it).`,
257
+ )
258
+ }
259
+ if (!sourceVideo && mode !== undefined) {
260
+ throw new Error(
261
+ `${this.name}: modelOptions.mode '${mode}' requires a video prompt ` +
262
+ `part carrying the source clip.`,
263
+ )
264
+ }
265
+
266
+ if (mode !== undefined && sourceVideo) {
267
+ return await this.createSourceVideoJob({
268
+ model,
269
+ mode,
270
+ sourceVideo,
271
+ resolved,
272
+ wireOptions,
273
+ size,
274
+ genericDuration: options.duration,
275
+ logger,
276
+ })
277
+ }
278
+
184
279
  validateVideoSize(model, size)
185
280
 
281
+ // Pull the specially-handled keys out of the wire options: `duration`
282
+ // is folded into the snapped value below, and the reference fields are
283
+ // re-added explicitly so a JSON-serialized `null` or empty array reads
284
+ // as "unset" instead of leaking onto the wire.
285
+ const {
286
+ duration: rawOptionDuration,
287
+ reference_images: explicitReferenceImages,
288
+ reference_audios: referenceAudios,
289
+ ...generationOptions
290
+ } = wireOptions
291
+
186
292
  // Coerce the requested duration into the model's valid range (1–15s,
187
293
  // integer) instead of rejecting it — `snapDuration` clamps and rounds.
188
294
  // modelOptions wins over the generic `duration`, mirroring the size
189
295
  // precedence below.
190
- const rawDuration = modelOptions?.duration ?? options.duration
296
+ const rawDuration = rawOptionDuration ?? options.duration
191
297
  const duration =
192
- rawDuration !== undefined ? this.snapDuration(rawDuration) : undefined
193
-
194
- // The interleaved prompt decomposes into verbatim text plus typed media
195
- // buckets. The Imagine video endpoint takes a text prompt and an optional
196
- // starting frame; reject the modalities it can't consume.
197
- const resolved = resolveMediaPrompt(options.prompt)
198
- if (resolved.videos.length > 0) {
298
+ rawDuration != null ? this.snapDuration(rawDuration) : undefined
299
+
300
+ // Image parts split by role: un-roled / 'start_frame' images become the
301
+ // starting frame (image-to-video); 'reference' / 'character' images
302
+ // become reference_images (reference-to-video). The Imagine API has no
303
+ // mask / control / end-frame inputs. Unknown role strings (possible via
304
+ // JSON callers) throw rather than silently dropping the part.
305
+ const startFrames: Array<ImagePart<MediaInputMetadata>> = []
306
+ const referenceImages: Array<{ url: string }> = []
307
+ for (const part of resolved.images) {
308
+ const role = part.metadata?.role
309
+ switch (role) {
310
+ case 'mask':
311
+ case 'control':
312
+ case 'end_frame':
313
+ throw new Error(
314
+ `${this.name}: the Imagine video API has no '${role}' image ` +
315
+ `input on model ${model}. Use an un-roled / 'start_frame' ` +
316
+ `image as the starting frame, or 'reference' images.`,
317
+ )
318
+ case 'reference':
319
+ case 'character':
320
+ referenceImages.push({ url: mediaPartToUrl(part) })
321
+ break
322
+ case 'start_frame':
323
+ case undefined:
324
+ startFrames.push(part)
325
+ break
326
+ default:
327
+ throw new Error(
328
+ `${this.name}: unknown image metadata.role '${String(role)}'. ` +
329
+ `Expected 'start_frame', 'reference', or 'character'.`,
330
+ )
331
+ }
332
+ }
333
+ if (startFrames.length > 1) {
199
334
  throw new Error(
200
- `${this.name}.createVideoJob does not support video prompt parts (model: ${model}).`,
335
+ `${this.name}: ${model} accepts at most one starting-frame image; received ${startFrames.length}. ` +
336
+ `Use metadata.role: 'reference' for reference-to-video inputs.`,
201
337
  )
202
338
  }
203
- if (resolved.audios.length > 0) {
339
+ // Explicit modelOptions.reference_images replaces the part-derived list
340
+ // (an explicit empty array means "none").
341
+ const finalReferenceImages =
342
+ explicitReferenceImages ??
343
+ (referenceImages.length > 0 ? referenceImages : undefined)
344
+ const referenceImageCount = finalReferenceImages?.length ?? 0
345
+ const referenceAudioCount = referenceAudios?.length ?? 0
346
+ const hasReference = referenceImageCount > 0 || referenceAudioCount > 0
347
+
348
+ // Reference inputs are a grok-imagine-video-1.5 feature. The per-model
349
+ // options map already hides the fields from other models at compile
350
+ // time; this runtime gate covers prompt-part roles and untyped callers.
351
+ if (!isGrokVideoReferenceModel(model) && hasReference) {
204
352
  throw new Error(
205
- `${this.name}.createVideoJob does not support audio prompt parts (model: ${model}).`,
353
+ `${this.name}: ${model} does not support reference-to-video inputs. ` +
354
+ `Use 'grok-imagine-video-1.5' for reference_images / reference_audios.`,
206
355
  )
207
356
  }
208
- // grok-imagine-video-1.5 is image-to-video only — text-to-video is
209
- // rejected by the API, so fail fast with a clear, actionable message
210
- // pointing at the model that does support text-to-video.
211
- if (resolved.images.length === 0 && isImageToVideoOnlyModel(model)) {
357
+ if (referenceAudioCount > GROK_VIDEO_MAX_REFERENCE_AUDIOS) {
212
358
  throw new Error(
213
- `${this.name}: ${model} does not support text-to-video — it is image-to-video only. ` +
214
- `Include an image prompt part as the starting frame, or use 'grok-imagine-video' for text-to-video.`,
359
+ `${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_AUDIOS} reference voices; received ${referenceAudioCount}.`,
215
360
  )
216
361
  }
217
- if (resolved.images.length > 1) {
362
+ if (referenceImageCount > GROK_VIDEO_MAX_REFERENCE_IMAGES) {
218
363
  throw new Error(
219
- `${this.name}: ${model} accepts at most one starting-frame image; received ${resolved.images.length}.`,
364
+ `${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_IMAGES} reference images; received ${referenceImageCount}.`,
220
365
  )
221
366
  }
222
367
 
223
368
  // Image-to-video: the single image prompt part becomes the starting frame
224
369
  // and the prompt text describes the desired motion. URL sources are
225
370
  // fetched by xAI's servers; data sources are sent as base64 data URIs.
226
- const [startFrame] = resolved.images
371
+ const [startFrame] = startFrames
372
+
373
+ // xAI rejects `image` + `reference_images` / `reference_audios` as a
374
+ // 400: only one of image-to-video or reference-to-video can be active.
375
+ if (startFrame && hasReference) {
376
+ throw new Error(
377
+ `${this.name}: image-to-video and reference-to-video cannot be combined. ` +
378
+ `Use a starting-frame image, or reference images / voices, not both.`,
379
+ )
380
+ }
227
381
 
228
382
  // The generic `size` option carries an "aspectRatio_resolution" template
229
383
  // (e.g. '16:9_720p') and maps to the Imagine API's `aspect_ratio` /
230
- // `resolution` parameters; explicit modelOptions win over the template.
384
+ // `resolution` parameters; explicit modelOptions win over the template
385
+ // (including `reference_images`, which replaces the part-derived list).
231
386
  const parsedSize = size !== undefined ? parseGrokVideoSize(size) : undefined
387
+ const resolvedResolution =
388
+ generationOptions.resolution ?? parsedSize?.resolution
389
+ if (hasReference && resolvedResolution === '1080p') {
390
+ throw new Error(
391
+ `${this.name}: reference-to-video is capped at 720p on ${model}.`,
392
+ )
393
+ }
232
394
  const request = {
233
395
  model,
234
396
  prompt: resolved.text,
235
- ...(startFrame && { image: { url: imagePartToUrl(startFrame) } }),
397
+ ...(startFrame && { image: { url: mediaPartToUrl(startFrame) } }),
398
+ ...(referenceImageCount > 0 && {
399
+ reference_images: finalReferenceImages,
400
+ }),
401
+ ...(referenceAudioCount > 0 && {
402
+ reference_audios: referenceAudios,
403
+ }),
236
404
  ...(parsedSize && {
237
405
  aspect_ratio: parsedSize.aspectRatio,
238
406
  ...(parsedSize.resolution !== undefined && {
239
407
  resolution: parsedSize.resolution,
240
408
  }),
241
409
  }),
242
- ...modelOptions,
243
- // Spread after modelOptions so the snapped duration is authoritative
244
- // (modelOptions.duration is folded into `duration` via snapDuration above).
410
+ // The remaining options spread after the size template so explicit
411
+ // aspect_ratio / resolution win over it; duration and the reference
412
+ // fields were destructured out above and re-added normalized.
413
+ ...generationOptions,
245
414
  ...(duration !== undefined && { duration }),
246
415
  }
247
416
 
248
- try {
249
- logger.request(
250
- `activity=video.create provider=${this.name} model=${model} size=${size ?? 'default'} duration=${duration ?? 'default'}`,
251
- { provider: this.name, model },
417
+ return await this.postVideoJob('/videos/generations', request, {
418
+ model,
419
+ logger,
420
+ logLine: `activity=video.create provider=${this.name} model=${model} mode=generate size=${size ?? 'default'} duration=${duration ?? 'default'}`,
421
+ })
422
+ }
423
+
424
+ /**
425
+ * Build and post an edit / extension request. Both endpoints take only
426
+ * `model`, `prompt`, and the source `video` (plus `duration` — the length
427
+ * of the added tail — for extensions): output geometry is inherited from
428
+ * the source clip, capped at 720p, and edit outputs also inherit the
429
+ * source length. Rather than sending fields the API documents as ignored,
430
+ * the inapplicable options are rejected with actionable errors.
431
+ */
432
+ private async createSourceVideoJob(args: {
433
+ model: string
434
+ mode: 'edit' | 'extend'
435
+ sourceVideo: VideoPart<MediaInputMetadata>
436
+ resolved: ReturnType<typeof resolveMediaPrompt>
437
+ wireOptions: Omit<GrokVideoRuntimeOptions, 'mode'>
438
+ size: string | undefined
439
+ genericDuration: number | undefined
440
+ logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']
441
+ }): Promise<VideoJobResult> {
442
+ const { model, mode, sourceVideo, resolved, wireOptions, logger } = args
443
+ const endpoint = mode === 'edit' ? '/videos/edits' : '/videos/extensions'
444
+
445
+ if (resolved.images.length > 0) {
446
+ throw new Error(
447
+ `${this.name}: '${mode}' mode takes only the source video — image ` +
448
+ `prompt parts are not supported by ${endpoint}.`,
252
449
  )
450
+ }
253
451
 
254
- const response = await this.request('/videos/generations', {
452
+ // Pull every generation-only key out of the wire options so nothing can
453
+ // leak into the edit/extend body via the spread below. JSON-serialized
454
+ // `null` values (a common "unset" encoding) are treated as absent;
455
+ // actual values are rejected with actionable errors.
456
+ const {
457
+ aspect_ratio: aspectRatio,
458
+ resolution,
459
+ duration: modeDuration,
460
+ reference_images: referenceImagesOption,
461
+ reference_audios: referenceAudiosOption,
462
+ ...passthrough
463
+ } = wireOptions
464
+ if (
465
+ (referenceImagesOption?.length ?? 0) > 0 ||
466
+ (referenceAudiosOption?.length ?? 0) > 0
467
+ ) {
468
+ throw new Error(
469
+ `${this.name}: reference inputs are only supported by video ` +
470
+ `generation, not '${mode}' mode.`,
471
+ )
472
+ }
473
+ if (args.size !== undefined || aspectRatio != null || resolution != null) {
474
+ throw new Error(
475
+ `${this.name}: '${mode}' mode does not accept size / aspect_ratio / ` +
476
+ `resolution — the output inherits the source clip's geometry ` +
477
+ `(capped at 720p).`,
478
+ )
479
+ }
480
+ const rawDuration = modeDuration ?? args.genericDuration
481
+ if (mode === 'edit' && rawDuration != null) {
482
+ throw new Error(
483
+ `${this.name}: 'edit' mode does not accept a duration — the output ` +
484
+ `inherits the source clip's length. Use mode 'extend' to append ` +
485
+ `seconds to the clip.`,
486
+ )
487
+ }
488
+ // Extend: the snapped duration is the added-tail length (1–15s).
489
+ const duration =
490
+ rawDuration != null ? this.snapDuration(rawDuration) : undefined
491
+
492
+ const request = {
493
+ model,
494
+ prompt: resolved.text,
495
+ video: { url: mediaPartToUrl(sourceVideo) },
496
+ ...passthrough,
497
+ ...(duration !== undefined && { duration }),
498
+ }
499
+
500
+ return await this.postVideoJob(endpoint, request, {
501
+ model,
502
+ logger,
503
+ logLine: `activity=video.create provider=${this.name} model=${model} mode=${mode} duration=${duration ?? 'default'}`,
504
+ })
505
+ }
506
+
507
+ /**
508
+ * POST a create-job request body to one of the Imagine video endpoints
509
+ * (`/videos/generations`, `/videos/edits`, `/videos/extensions`) and read
510
+ * the `request_id` out of the shared response shape.
511
+ */
512
+ private async postVideoJob(
513
+ endpoint: string,
514
+ request: Record<string, unknown>,
515
+ context: {
516
+ model: string
517
+ logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']
518
+ logLine: string
519
+ },
520
+ ): Promise<VideoJobResult> {
521
+ const { model, logger, logLine } = context
522
+ try {
523
+ logger.request(logLine, { provider: this.name, model })
524
+
525
+ const response = await this.request(endpoint, {
255
526
  method: 'POST',
256
527
  body: JSON.stringify(request),
257
528
  })
258
529
  if (!response.ok) {
259
530
  throw new Error(
260
- `grok: video generation request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,
531
+ `grok: ${endpoint} request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,
261
532
  )
262
533
  }
263
534
 
264
535
  const result = (await response.json()) as GrokVideoCreateResponse
265
536
  if (!result.request_id) {
266
- throw new Error(
267
- 'grok: video generation response contained no request_id',
268
- )
537
+ throw new Error(`grok: ${endpoint} response contained no request_id`)
269
538
  }
270
539
  return { jobId: result.request_id, model }
271
540
  } catch (error: unknown) {
@@ -394,15 +663,14 @@ export class GrokVideoAdapter<
394
663
  *
395
664
  * @experimental Video generation is an experimental feature and may change.
396
665
  *
397
- * @param model - The model name (e.g., 'grok-imagine-video')
666
+ * @param model - The model name (e.g., 'grok-imagine-video-1.5')
398
667
  * @param apiKey - Your xAI API key
399
668
  * @param config - Optional additional configuration
400
669
  * @returns Configured Grok video adapter instance with resolved types
401
670
  *
402
671
  * @example
403
672
  * ```typescript
404
- * // grok-imagine-video (v1.0) supports text-to-video.
405
- * const adapter = createGrokVideo('grok-imagine-video', 'xai-...');
673
+ * const adapter = createGrokVideo('grok-imagine-video-1.5', 'xai-...');
406
674
  *
407
675
  * const { jobId } = await generateVideo({
408
676
  * adapter,
@@ -440,7 +708,7 @@ export function createGrokVideo<TModel extends GrokVideoModel>(
440
708
  * // Automatically uses XAI_API_KEY from environment
441
709
  * const adapter = grokVideo('grok-imagine-video-1.5');
442
710
  *
443
- * // Image-to-video only: the prompt must carry a starting-frame image part.
711
+ * // Image-to-video: an optional image prompt part is the starting frame.
444
712
  * const { jobId } = await generateVideo({
445
713
  * adapter,
446
714
  * prompt: [
@@ -1,8 +1,9 @@
1
1
  /**
2
2
  * Grok Image Generation Provider Options
3
3
  *
4
- * These are provider-specific options for Grok image generation.
5
- * Grok uses the grok-2-image-1212 model for image generation.
4
+ * Provider-specific options for Grok image generation: the aspect-ratio
5
+ * sized Imagine API models (grok-imagine-image, grok-imagine-image-2.0,
6
+ * grok-imagine-image-quality) and the legacy pixel-sized grok-2-image-1212.
6
7
  */
7
8
 
8
9
  /**
@@ -141,12 +142,25 @@ export interface GrokImagineImageProviderOptions extends GrokImageBaseProviderOp
141
142
  service_tier?: 'default' | 'priority'
142
143
  }
143
144
 
145
+ /**
146
+ * Provider options for grok-imagine-image-2.0, which adds a generation
147
+ * `quality` knob on top of the shared Imagine options.
148
+ */
149
+ export interface GrokImagineImage2ProviderOptions extends GrokImagineImageProviderOptions {
150
+ /**
151
+ * Generation quality. Only supported by grok-imagine-image-2.0.
152
+ * @default 'medium'
153
+ */
154
+ quality?: 'low' | 'medium'
155
+ }
156
+
144
157
  /**
145
158
  * Type-only map from model name to its specific provider options.
146
159
  */
147
160
  export type GrokImageModelProviderOptionsByName = {
148
161
  'grok-2-image-1212': GrokImageProviderOptions
149
162
  'grok-imagine-image': GrokImagineImageProviderOptions
163
+ 'grok-imagine-image-2.0': GrokImagineImage2ProviderOptions
150
164
  'grok-imagine-image-quality': GrokImagineImageProviderOptions
151
165
  }
152
166
 
@@ -156,6 +170,7 @@ export type GrokImageModelProviderOptionsByName = {
156
170
  export type GrokImageModelSizeByName = {
157
171
  'grok-2-image-1212': GrokImageSize
158
172
  'grok-imagine-image': GrokImagineImageSize
173
+ 'grok-imagine-image-2.0': GrokImagineImageSize
159
174
  'grok-imagine-image-quality': GrokImagineImageSize
160
175
  }
161
176
 
@@ -167,6 +182,7 @@ export type GrokImageModelSizeByName = {
167
182
  export type GrokImageModelInputModalitiesByName = {
168
183
  'grok-2-image-1212': readonly []
169
184
  'grok-imagine-image': readonly ['image']
185
+ 'grok-imagine-image-2.0': readonly ['image']
170
186
  'grok-imagine-image-quality': readonly ['image']
171
187
  }
172
188
 
package/src/index.ts CHANGED
@@ -28,6 +28,8 @@ export {
28
28
  } from './adapters/image'
29
29
  export type {
30
30
  GrokImageProviderOptions,
31
+ GrokImagineImageProviderOptions,
32
+ GrokImagineImage2ProviderOptions,
31
33
  GrokImageModelProviderOptionsByName,
32
34
  } from './image/image-provider-options'
33
35
 
@@ -43,7 +45,11 @@ export {
43
45
  getGrokVideoDurationOptions,
44
46
  } from './video/video-provider-options'
45
47
  export type {
48
+ GrokVideoMode,
49
+ GrokVideoBaseProviderOptions,
50
+ GrokVideoSourceProviderOptions,
46
51
  GrokVideoProviderOptions,
52
+ GrokVideoRuntimeOptions,
47
53
  GrokVideoModelProviderOptionsByName,
48
54
  GrokVideoModelSizeByName,
49
55
  GrokVideoModelDurationByName,
@@ -101,6 +107,7 @@ export {
101
107
  GROK_TTS_MODELS,
102
108
  GROK_TRANSCRIPTION_MODELS,
103
109
  GROK_REALTIME_MODELS,
110
+ GROK_DEFAULT_REALTIME_MODEL,
104
111
  } from './model-meta'
105
112
  export type {
106
113
  GrokTextMetadata,