@ai-sdk/xai 3.0.120 → 3.0.122

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/docs/01-xai.mdx CHANGED
@@ -1273,12 +1273,64 @@ const { video } = await generateVideo({
1273
1273
  });
1274
1274
  ```
1275
1275
 
1276
+ `inputReferences` accepts image references only. A reference with a non-image
1277
+ media type (for example a video or audio clip) is ignored with a warning, and a
1278
+ reference with no media type is treated as an image. If no image reference
1279
+ remains, reference-to-video is not selected and no reference images are sent.
1280
+
1281
+ #### Reference Audio
1282
+
1283
+ Reference-to-video can also give the subject a voice. `referenceVoiceIds` takes
1284
+ up to 3 xAI **preset** voice ids — you cannot upload your own audio clips.
1285
+ Reference the voices from the prompt with `<AUDIO_0>`, `<AUDIO_1>`, and
1286
+ `<AUDIO_2>`, in the order the voices are passed.
1287
+
1288
+ ```ts
1289
+ import { xai, type XaiVideoModelOptions } from '@ai-sdk/xai';
1290
+ import { experimental_generateVideo as generateVideo } from 'ai';
1291
+
1292
+ const { video } = await generateVideo({
1293
+ model: xai.video('grok-imagine-video-1.5'),
1294
+ prompt:
1295
+ 'The person from <IMAGE_1> stands in the room from <IMAGE_2> and speaks ' +
1296
+ 'to the camera with the voice from <AUDIO_0>.',
1297
+ aspectRatio: '9:16',
1298
+ duration: 10,
1299
+ providerOptions: {
1300
+ xai: {
1301
+ mode: 'reference-to-video',
1302
+ referenceImageUrls: [
1303
+ 'https://example.com/person.png',
1304
+ 'https://example.com/room.png',
1305
+ ],
1306
+ referenceVoiceIds: ['eve'],
1307
+ resolution: '720p',
1308
+ pollTimeoutMs: 600000,
1309
+ } satisfies XaiVideoModelOptions,
1310
+ },
1311
+ });
1312
+ ```
1313
+
1314
+ Valid voice ids come from the xAI
1315
+ [text-to-speech voice roster](https://docs.x.ai/developers/model-capabilities/audio/text-to-speech#voices).
1316
+ Ids are case-insensitive, and an unknown id returns a `400` response listing the
1317
+ available voices. `referenceVoiceIds` is ignored with a warning when the
1318
+ resolved operation is not reference-to-video.
1319
+
1320
+ <Note>
1321
+ xAI documents reference audio as available in the United States only, for
1322
+ trusted partners. Requests from accounts without access are rejected by the
1323
+ xAI API.
1324
+ </Note>
1325
+
1276
1326
  <Note>
1277
1327
  Reference-to-video supports `duration`, `aspectRatio`, and `resolution`. Use
1278
1328
  `mode` to select the operation — each mode is mutually exclusive. When both
1279
1329
  are provided, `frameImages` takes precedence over `inputReferences`, and
1280
1330
  `inputReferences` takes precedence over the legacy `referenceImageUrls`
1281
- provider option. Reference-to-video requires the `grok-imagine-video` model.
1331
+ provider option. Reference-to-video is supported by the `grok-imagine-video`
1332
+ and `grok-imagine-video-1.5` models; native 1080p requires
1333
+ `grok-imagine-video-1.5`.
1282
1334
  </Note>
1283
1335
 
1284
1336
  ### Video Provider Options
@@ -1294,11 +1346,14 @@ You can validate the provider options using the `XaiVideoModelOptions` type.
1294
1346
 
1295
1347
  Maximum wait time in milliseconds for video generation. Defaults to 600000 (10 minutes).
1296
1348
 
1297
- - **resolution** _'480p' | '720p'_
1349
+ - **resolution** _'480p' | '720p' | '1080p'_
1298
1350
 
1299
1351
  Video resolution. When using the SDK's standard `resolution` parameter,
1300
- `1280x720` maps to `720p` and `854x480` maps to `480p`.
1301
- Use this provider option to pass the native format directly.
1352
+ `1920x1080` maps to `1080p`, `1280x720` maps to `720p`, and `854x480` maps
1353
+ to `480p`. Use this provider option to pass the native format directly.
1354
+ `1080p` requires the `grok-imagine-video-1.5` model and is available for
1355
+ text-to-video and image-to-video; reference-to-video is capped at `720p`
1356
+ (a `1080p` request is downgraded with a warning).
1302
1357
 
1303
1358
  - **user** _string_
1304
1359
 
@@ -1330,6 +1385,17 @@ You can validate the provider options using the `XaiVideoModelOptions` type.
1330
1385
  `<IMAGE_1>`, `<IMAGE_2>`, etc. in the prompt to reference specific
1331
1386
  images. Used with `mode: 'reference-to-video'`.
1332
1387
 
1388
+ - **referenceVoiceIds** _string[]_
1389
+
1390
+ Up to 3 xAI preset voice ids that give the subject a voice in
1391
+ reference-to-video (R2V) generation. Preset voices only — audio clips
1392
+ cannot be uploaded. Ids are case-insensitive and come from the
1393
+ [text-to-speech voice roster](https://docs.x.ai/developers/model-capabilities/audio/text-to-speech#voices);
1394
+ an unknown id returns a `400` listing the available voices. Use `<AUDIO_0>`,
1395
+ `<AUDIO_1>`, and `<AUDIO_2>` tags in the prompt to reference the voices.
1396
+ Ignored with a warning outside reference-to-video. Reference audio is
1397
+ documented as US-only and limited to trusted partners.
1398
+
1333
1399
  <Note>
1334
1400
  Video generation is an asynchronous process that can take several minutes.
1335
1401
  Consider setting `pollTimeoutMs` to at least 10 minutes (600000ms) for
@@ -1340,7 +1406,9 @@ You can validate the provider options using the `XaiVideoModelOptions` type.
1340
1406
  ### Aspect Ratio and Resolution
1341
1407
 
1342
1408
  For **text-to-video**, you can specify both `aspectRatio` and `resolution`.
1343
- The default aspect ratio is `16:9` and the default resolution is `480p`.
1409
+ The default aspect ratio is `16:9` and the default resolution is `480p`. The
1410
+ `grok-imagine-video-1.5` model additionally supports native `1080p` for
1411
+ text-to-video and image-to-video.
1344
1412
 
1345
1413
  For **image-to-video**, the output defaults to the input image's aspect ratio.
1346
1414
  If you specify `aspectRatio`, it will override this and stretch the image to the
@@ -1356,13 +1424,18 @@ from the source video. `duration` is supported and controls only the
1356
1424
  extension length.
1357
1425
 
1358
1426
  For **reference-to-video (R2V)**, you can specify `duration`, `aspectRatio`,
1359
- and `resolution` just like text-to-video.
1427
+ and `resolution`. Unlike text-to-video, R2V is capped at `720p` — a `1080p`
1428
+ request is downgraded to `720p` with a warning.
1360
1429
 
1361
1430
  ### Video Model Capabilities
1362
1431
 
1363
- | Model | Duration | Aspect Ratios | Resolution | Image-to-Video | Editing | Extension | R2V |
1364
- | -------------------- | -------- | ------------------------------------------------- | -------------- | ------------------- | ------------------- | ------------------- | ------------------- |
1365
- | `grok-imagine-video` | 1–15s | `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `3:2`, `2:3` | `480p`, `720p` | <Check size={18} /> | <Check size={18} /> | <Check size={18} /> | <Check size={18} /> |
1432
+ | Model | Duration | Aspect Ratios | Resolution | Image-to-Video | Editing | Extension | R2V |
1433
+ | ------------------------ | -------- | ------------------------------------------------- | ------------------------- | ------------------- | ------------------- | ------------------- | ------------------- |
1434
+ | `grok-imagine-video` | 1–15s | `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `3:2`, `2:3` | `480p`, `720p` | <Check size={18} /> | <Check size={18} /> | <Check size={18} /> | <Check size={18} /> |
1435
+ | `grok-imagine-video-1.5` | 1–15s | `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `3:2`, `2:3` | `480p`, `720p`, `1080p`\* | <Check size={18} /> | <Check size={18} /> | <Check size={18} /> | <Check size={18} /> |
1436
+
1437
+ \* Native `1080p` applies to text-to-video and image-to-video. Reference-to-video
1438
+ is capped at `720p` — a `1080p` request is downgraded with a warning.
1366
1439
 
1367
1440
  <Note>
1368
1441
  You can also pass any available provider model ID as a string if needed.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/xai",
3
- "version": "3.0.120",
3
+ "version": "3.0.122",
4
4
  "license": "Apache-2.0",
5
5
  "sideEffects": false,
6
6
  "main": "./dist/index.js",
@@ -29,17 +29,17 @@
29
29
  }
30
30
  },
31
31
  "dependencies": {
32
- "@ai-sdk/openai-compatible": "2.0.67",
33
- "@ai-sdk/provider-utils": "4.0.45",
34
- "@ai-sdk/provider": "3.0.15"
32
+ "@ai-sdk/openai-compatible": "2.0.68",
33
+ "@ai-sdk/provider": "3.0.15",
34
+ "@ai-sdk/provider-utils": "4.0.46"
35
35
  },
36
36
  "devDependencies": {
37
37
  "@types/node": "20.17.24",
38
38
  "tsup": "^8",
39
39
  "typescript": "5.8.3",
40
40
  "zod": "3.25.76",
41
- "@ai-sdk/test-server": "1.0.6",
42
- "@vercel/ai-tsconfig": "0.0.0"
41
+ "@vercel/ai-tsconfig": "0.0.0",
42
+ "@ai-sdk/test-server": "1.0.6"
43
43
  },
44
44
  "peerDependencies": {
45
45
  "zod": "^3.25.76 || ^4.1.8"
@@ -37,6 +37,7 @@ interface XaiVideoModelConfig {
37
37
  }
38
38
 
39
39
  const RESOLUTION_MAP: Record<string, string> = {
40
+ '1920x1080': '1080p',
40
41
  '1280x720': '720p',
41
42
  '854x480': '480p',
42
43
  '640x480': '480p',
@@ -74,6 +75,11 @@ function getTopLevelMediaType(mediaType: string): string {
74
75
  const isVideoFile = (file: Experimental_VideoModelV3File): boolean =>
75
76
  file.mediaType != null && getTopLevelMediaType(file.mediaType) === 'video';
76
77
 
78
+ // References without a media type (only possible for URLs) are treated as
79
+ // images, matching the legacy `referenceImageUrls` behavior.
80
+ const isImageReference = (file: Experimental_VideoModelV3File): boolean =>
81
+ file.mediaType == null || getTopLevelMediaType(file.mediaType) === 'image';
82
+
77
83
  function fileToXaiImageUrl(file: Experimental_VideoModelV3File): string {
78
84
  if (file.type === 'url') {
79
85
  return file.url;
@@ -88,32 +94,40 @@ function fileToXaiImageUrl(file: Experimental_VideoModelV3File): string {
88
94
 
89
95
  // Resolves the reference images for R2V generation. First-class
90
96
  // `inputReferences` win over the legacy `referenceImageUrls` provider option.
91
- // Video references are not supported for reference-to-video and are skipped
92
- // with a warning.
97
+ // Non-image references (video or audio) are not supported for
98
+ // reference-to-video and are skipped with a warning.
93
99
  function resolveReferenceImages(
94
100
  options: XaiVideoDoGenerateOptions,
95
101
  xaiOptions: XaiParsedVideoModelOptions | undefined,
96
102
  warnings: SharedV3Warning[],
97
103
  ): Array<{ url: string }> | undefined {
98
104
  if (options.inputReferences != null && options.inputReferences.length > 0) {
99
- const imageReferences = options.inputReferences.filter(reference => {
100
- if (isVideoFile(reference)) {
105
+ const imageFiles: Experimental_VideoModelV3File[] = [];
106
+
107
+ for (const reference of options.inputReferences) {
108
+ if (!isImageReference(reference)) {
101
109
  warnings.push({
102
110
  type: 'unsupported',
103
111
  feature: 'inputReferences',
104
- details:
105
- 'xAI reference-to-video accepts image references only. The video ' +
106
- 'reference was ignored. Use providerOptions.xai.mode ' +
107
- '"extend-video" to continue from a video.',
112
+ details: isVideoFile(reference)
113
+ ? 'xAI reference-to-video accepts image references only. The ' +
114
+ 'video reference was ignored. Use providerOptions.xai.mode ' +
115
+ '"extend-video" to continue from a video.'
116
+ : 'xAI reference-to-video accepts image references only. The ' +
117
+ 'non-image reference was ignored.',
108
118
  });
109
- return false;
119
+ continue;
110
120
  }
111
- return true;
112
- });
113
121
 
114
- return imageReferences.map(reference => ({
115
- url: fileToXaiImageUrl(reference),
116
- }));
122
+ imageFiles.push(reference);
123
+ }
124
+
125
+ // Every reference may have been filtered out (audio- or video-only input),
126
+ // so collapse an empty list to undefined rather than sending an empty
127
+ // `reference_images` array.
128
+ return imageFiles.length > 0
129
+ ? imageFiles.map(reference => ({ url: fileToXaiImageUrl(reference) }))
130
+ : undefined;
117
131
  }
118
132
 
119
133
  if (
@@ -126,6 +140,11 @@ function resolveReferenceImages(
126
140
  return undefined;
127
141
  }
128
142
 
143
+ // True when at least one reference would survive as an image.
144
+ function hasImageInputReference(options: XaiVideoDoGenerateOptions): boolean {
145
+ return options.inputReferences?.some(isImageReference) ?? false;
146
+ }
147
+
129
148
  function resolveVideoMode(
130
149
  options: XaiVideoDoGenerateOptions,
131
150
  xaiOptions: XaiParsedVideoModelOptions | undefined,
@@ -142,13 +161,17 @@ function resolveVideoMode(
142
161
  // only auto-select reference-to-video when no frame images are provided.
143
162
  const hasFrameImages =
144
163
  options.frameImages != null && options.frameImages.length > 0;
145
- const hasInputReferences =
146
- options.inputReferences != null && options.inputReferences.length > 0;
147
164
  const hasLegacyReferenceUrls =
148
165
  xaiOptions?.referenceImageUrls != null &&
149
166
  xaiOptions.referenceImageUrls.length > 0;
150
167
 
151
- if (!hasFrameImages && (hasInputReferences || hasLegacyReferenceUrls)) {
168
+ // Reference-to-video needs at least one image reference. An audio-only (or
169
+ // video-only) `inputReferences` array must not flip a text- or
170
+ // image-to-video request into R2V.
171
+ if (
172
+ !hasFrameImages &&
173
+ (hasImageInputReference(options) || hasLegacyReferenceUrls)
174
+ ) {
152
175
  return 'reference-to-video';
153
176
  }
154
177
 
@@ -289,7 +312,8 @@ export class XaiVideoModel implements Experimental_VideoModelV3 {
289
312
  feature: 'resolution',
290
313
  details:
291
314
  `Unrecognized resolution "${options.resolution}". ` +
292
- 'Use providerOptions.xai.resolution with "480p" or "720p" instead.',
315
+ 'Use providerOptions.xai.resolution with "480p", "720p", or ' +
316
+ '"1080p" instead.',
293
317
  });
294
318
  }
295
319
  }
@@ -346,13 +370,57 @@ export class XaiVideoModel implements Experimental_VideoModelV3 {
346
370
  xaiOptions,
347
371
  warnings,
348
372
  );
373
+
349
374
  if (referenceImages != null) {
350
375
  body.reference_images = referenceImages;
376
+ } else {
377
+ // Explicit R2V with no usable image references would silently send
378
+ // a plain generations request; tell the user it is no longer R2V.
379
+ warnings.push({
380
+ type: 'unsupported',
381
+ feature: 'referenceImages',
382
+ details:
383
+ 'xAI reference-to-video requires at least one image reference. ' +
384
+ 'The video will be generated without reference images.',
385
+ });
386
+ }
387
+
388
+ const referenceVoiceIds = xaiOptions?.referenceVoiceIds;
389
+ if (referenceVoiceIds != null && referenceVoiceIds.length > 0) {
390
+ body.reference_audios = referenceVoiceIds.map(voiceId => ({
391
+ voice_id: voiceId,
392
+ }));
393
+ }
394
+
395
+ // Reference-to-video is capped at 720p; downgrade a 1080p request.
396
+ if (body.resolution === '1080p') {
397
+ warnings.push({
398
+ type: 'unsupported',
399
+ feature: 'resolution',
400
+ details:
401
+ 'xAI reference-to-video is limited to 720p. The request was ' +
402
+ 'downgraded from 1080p to 720p.',
403
+ });
404
+ body.resolution = '720p';
351
405
  }
352
406
  }
353
407
 
354
- // Warn when reference images were provided but cannot be used in the
355
- // resolved mode (e.g. alongside frameImages, or in edit/extend modes).
408
+ // 1080p requires grok-imagine-video-1.5; the original grok-imagine-video
409
+ // rejects it. Warn, but send the request as the user asked.
410
+ if (body.resolution === '1080p' && this.modelId === 'grok-imagine-video') {
411
+ warnings.push({
412
+ type: 'unsupported',
413
+ feature: 'resolution',
414
+ details:
415
+ 'xAI model "grok-imagine-video" does not support 1080p. Use ' +
416
+ '"grok-imagine-video-1.5" for 1080p, or a lower resolution. The ' +
417
+ 'request was sent with 1080p.',
418
+ });
419
+ }
420
+
421
+ // Warn when references were provided but cannot be used in the resolved
422
+ // mode (e.g. alongside frameImages, in edit/extend modes, or when the
423
+ // references carried no usable image to drive reference-to-video).
356
424
  if (
357
425
  options.inputReferences != null &&
358
426
  options.inputReferences.length > 0 &&
@@ -361,9 +429,26 @@ export class XaiVideoModel implements Experimental_VideoModelV3 {
361
429
  warnings.push({
362
430
  type: 'unsupported',
363
431
  feature: 'inputReferences',
432
+ details: hasImageInputReference(options)
433
+ ? 'xAI only supports inputReferences for reference-to-video ' +
434
+ 'generation. The reference images were ignored.'
435
+ : 'xAI reference-to-video requires at least one image reference. ' +
436
+ 'The references were ignored.',
437
+ });
438
+ }
439
+
440
+ // Preset reference voices only apply to reference-to-video generation.
441
+ if (
442
+ xaiOptions?.referenceVoiceIds != null &&
443
+ xaiOptions.referenceVoiceIds.length > 0 &&
444
+ !hasReferenceImages
445
+ ) {
446
+ warnings.push({
447
+ type: 'unsupported',
448
+ feature: 'referenceVoiceIds',
364
449
  details:
365
- 'xAI only supports inputReferences for reference-to-video ' +
366
- 'generation. The reference images were ignored.',
450
+ 'xAI only supports reference voices for reference-to-video ' +
451
+ 'generation. The reference voices were ignored.',
367
452
  });
368
453
  }
369
454
 
@@ -381,6 +466,7 @@ export class XaiVideoModel implements Experimental_VideoModelV3 {
381
466
  'resolution',
382
467
  'videoUrl',
383
468
  'referenceImageUrls',
469
+ 'referenceVoiceIds',
384
470
  'user',
385
471
  ].includes(key)
386
472
  ) {
@@ -2,7 +2,7 @@ import { lazySchema, zodSchema } from '@ai-sdk/provider-utils';
2
2
  import { z } from 'zod/v4';
3
3
 
4
4
  const nonEmptyStringSchema = z.string().min(1);
5
- const resolutionSchema = z.enum(['480p', '720p']);
5
+ const resolutionSchema = z.enum(['480p', '720p', '1080p']);
6
6
  const modeSchema = z.enum(['edit-video', 'extend-video', 'reference-to-video']);
7
7
 
8
8
  export type XaiVideoMode = z.infer<typeof modeSchema>;
@@ -48,6 +48,10 @@ interface XaiVideoReferenceToVideoOptions
48
48
  mode: 'reference-to-video';
49
49
  /** Reference image URLs (1-7) for R2V generation. */
50
50
  referenceImageUrls: string[];
51
+ /**
52
+ * Preset voice ids (up to 3) that give the subject a voice.
53
+ */
54
+ referenceVoiceIds?: string[];
51
55
  }
52
56
 
53
57
  interface XaiVideoGenerationOptions
@@ -75,6 +79,10 @@ interface XaiLegacyReferenceToVideoOptions
75
79
  */
76
80
  mode?: undefined;
77
81
  referenceImageUrls: string[];
82
+ /**
83
+ * Preset voice ids (up to 3) that give the subject a voice.
84
+ */
85
+ referenceVoiceIds?: string[];
78
86
  }
79
87
 
80
88
  /**
@@ -87,6 +95,9 @@ interface XaiLegacyReferenceToVideoOptions
87
95
  * - `'reference-to-video'` + `referenceImageUrls` -- R2V generation (`POST /v1/videos/generations`)
88
96
  * - no `mode` -- standard generation from text prompts or image input
89
97
  *
98
+ * Reference images may also come from the top-level `inputReferences` option
99
+ * instead of `referenceImageUrls`.
100
+ *
90
101
  * Runtime remains backward compatible with legacy auto-detected provider
91
102
  * options, but the public TypeScript type is intentionally explicit so editors
92
103
  * can suggest valid modes and flag invalid field combinations.
@@ -110,12 +121,17 @@ const userField = {
110
121
  user: z.string().optional(),
111
122
  };
112
123
 
124
+ const referenceVoiceIdsField = {
125
+ referenceVoiceIds: z.array(nonEmptyStringSchema).max(3).optional(),
126
+ };
127
+
113
128
  const editVideoSchema = z.object({
114
129
  ...baseFields,
115
130
  ...userField,
116
131
  mode: z.literal('edit-video'),
117
132
  videoUrl: nonEmptyStringSchema,
118
133
  referenceImageUrls: z.undefined().optional(),
134
+ referenceVoiceIds: z.undefined().optional(),
119
135
  });
120
136
 
121
137
  const extendVideoSchema = z.object({
@@ -123,11 +139,13 @@ const extendVideoSchema = z.object({
123
139
  mode: z.literal('extend-video'),
124
140
  videoUrl: nonEmptyStringSchema,
125
141
  referenceImageUrls: z.undefined().optional(),
142
+ referenceVoiceIds: z.undefined().optional(),
126
143
  });
127
144
 
128
145
  const referenceToVideoSchema = z.object({
129
146
  ...baseFields,
130
147
  ...userField,
148
+ ...referenceVoiceIdsField,
131
149
  mode: z.literal('reference-to-video'),
132
150
  referenceImageUrls: z.array(nonEmptyStringSchema).min(1).max(7),
133
151
  videoUrl: z.undefined().optional(),
@@ -136,6 +154,7 @@ const referenceToVideoSchema = z.object({
136
154
  const autoDetectSchema = z.object({
137
155
  ...baseFields,
138
156
  ...userField,
157
+ ...referenceVoiceIdsField,
139
158
  mode: z.undefined().optional(),
140
159
  videoUrl: nonEmptyStringSchema.optional(),
141
160
  referenceImageUrls: z.array(nonEmptyStringSchema).min(1).max(7).optional(),
@@ -153,6 +172,7 @@ const runtimeSchema = z
153
172
  mode: modeSchema.optional(),
154
173
  videoUrl: nonEmptyStringSchema.optional(),
155
174
  referenceImageUrls: z.array(nonEmptyStringSchema).min(1).max(7).optional(),
175
+ referenceVoiceIds: z.array(nonEmptyStringSchema).max(3).optional(),
156
176
  user: z.string().optional(),
157
177
  ...baseFields,
158
178
  })
@@ -1 +1,4 @@
1
- export type XaiVideoModelId = 'grok-imagine-video' | (string & {});
1
+ export type XaiVideoModelId =
2
+ | 'grok-imagine-video'
3
+ | 'grok-imagine-video-1.5'
4
+ | (string & {});