@tanstack/ai-grok 0.14.10 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.js +1 -1
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/tts.js +2 -1
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/adapters/video.d.ts +33 -11
- package/dist/esm/adapters/video.js +125 -27
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/image/image-provider-options.d.ts +17 -2
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +3 -3
- package/dist/esm/index.js +2 -2
- package/dist/esm/model-meta.d.ts +49 -3
- package/dist/esm/model-meta.js +110 -8
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.js +17 -16
- package/dist/esm/realtime/adapter.js.map +1 -1
- package/dist/esm/realtime/token.d.ts +1 -1
- package/dist/esm/realtime/token.js +3 -2
- package/dist/esm/realtime/token.js.map +1 -1
- package/dist/esm/realtime/types.d.ts +1 -1
- package/dist/esm/tools/index.js.map +1 -1
- package/dist/esm/video/video-provider-options.d.ts +131 -21
- package/dist/esm/video/video-provider-options.js +36 -10
- package/dist/esm/video/video-provider-options.js.map +1 -1
- package/package.json +7 -7
- package/src/adapters/image.ts +2 -1
- package/src/adapters/video.ts +316 -51
- package/src/image/image-provider-options.ts +18 -2
- package/src/index.ts +7 -0
- package/src/model-meta.ts +109 -6
- package/src/realtime/adapter.ts +3 -2
- package/src/realtime/token.ts +4 -2
- package/src/realtime/types.ts +1 -1
- package/src/video/video-provider-options.ts +198 -34
package/src/model-meta.ts
CHANGED
|
@@ -29,6 +29,46 @@ interface ModelMeta {
|
|
|
29
29
|
}
|
|
30
30
|
}
|
|
31
31
|
|
|
32
|
+
const GROK_4_5 = {
|
|
33
|
+
name: 'grok-4.5',
|
|
34
|
+
context_window: 500_000,
|
|
35
|
+
supports: {
|
|
36
|
+
input: ['text', 'image', 'document'],
|
|
37
|
+
output: ['text'],
|
|
38
|
+
capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
|
|
39
|
+
tools: [],
|
|
40
|
+
},
|
|
41
|
+
pricing: {
|
|
42
|
+
input: {
|
|
43
|
+
normal: 2,
|
|
44
|
+
cached: 0.3,
|
|
45
|
+
},
|
|
46
|
+
output: {
|
|
47
|
+
normal: 6,
|
|
48
|
+
},
|
|
49
|
+
},
|
|
50
|
+
} as const satisfies ModelMeta
|
|
51
|
+
|
|
52
|
+
const GROK_4_6 = {
|
|
53
|
+
name: 'grok-4.6',
|
|
54
|
+
context_window: 500_000,
|
|
55
|
+
supports: {
|
|
56
|
+
input: ['text', 'image', 'document'],
|
|
57
|
+
output: ['text'],
|
|
58
|
+
capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
|
|
59
|
+
tools: [],
|
|
60
|
+
},
|
|
61
|
+
pricing: {
|
|
62
|
+
input: {
|
|
63
|
+
normal: 2,
|
|
64
|
+
cached: 0.5,
|
|
65
|
+
},
|
|
66
|
+
output: {
|
|
67
|
+
normal: 6,
|
|
68
|
+
},
|
|
69
|
+
},
|
|
70
|
+
} as const satisfies ModelMeta
|
|
71
|
+
|
|
32
72
|
export type GrokProviderToolKind =
|
|
33
73
|
| 'web_search'
|
|
34
74
|
| 'x_search'
|
|
@@ -91,17 +131,38 @@ const GROK_IMAGINE_IMAGE_QUALITY = {
|
|
|
91
131
|
},
|
|
92
132
|
} as const satisfies ModelMeta
|
|
93
133
|
|
|
134
|
+
// xAI's recommended Imagine image model. Supports the 2.0-only `quality`
|
|
135
|
+
// provider option ('low' | 'medium', default 'medium').
|
|
136
|
+
const GROK_IMAGINE_IMAGE_2_0 = {
|
|
137
|
+
name: 'grok-imagine-image-2.0',
|
|
138
|
+
supports: {
|
|
139
|
+
input: ['text', 'image'],
|
|
140
|
+
output: ['image'],
|
|
141
|
+
},
|
|
142
|
+
pricing: {
|
|
143
|
+
input: {
|
|
144
|
+
normal: 0,
|
|
145
|
+
},
|
|
146
|
+
output: {
|
|
147
|
+
normal: 0.04,
|
|
148
|
+
},
|
|
149
|
+
},
|
|
150
|
+
} as const satisfies ModelMeta
|
|
151
|
+
|
|
94
152
|
// Imagine API video models. Pricing is per second of generated video
|
|
95
153
|
// (output only); generated videos carry an audio track.
|
|
96
154
|
//
|
|
97
|
-
//
|
|
98
|
-
// optional)
|
|
99
|
-
//
|
|
100
|
-
//
|
|
155
|
+
// Both models support text-to-video and image-to-video (a starting-frame
|
|
156
|
+
// image is optional). grok-imagine-video-1.5 is the documented default: it
|
|
157
|
+
// adds native 1080p for text-to-video / image-to-video plus
|
|
158
|
+
// reference-to-video (`reference_images` / `reference_audios`; reference
|
|
159
|
+
// output is capped at 720p). Source-video edit (`/v1/videos/edits`) and
|
|
160
|
+
// extend (`/v1/videos/extensions`) are grok-imagine-video only — xAI's
|
|
161
|
+
// 1.5 model page lists text+image input, not video.
|
|
101
162
|
const GROK_IMAGINE_VIDEO = {
|
|
102
163
|
name: 'grok-imagine-video',
|
|
103
164
|
supports: {
|
|
104
|
-
input: ['text', 'image'],
|
|
165
|
+
input: ['text', 'image', 'video'],
|
|
105
166
|
output: ['video', 'audio'],
|
|
106
167
|
},
|
|
107
168
|
pricing: {
|
|
@@ -175,7 +236,12 @@ const GROK_BUILD_0_1 = {
|
|
|
175
236
|
/**
|
|
176
237
|
* Grok chat models supported by the Responses adapter.
|
|
177
238
|
*/
|
|
178
|
-
export const GROK_CHAT_MODELS = [
|
|
239
|
+
export const GROK_CHAT_MODELS = [
|
|
240
|
+
GROK_4_5.name,
|
|
241
|
+
GROK_4_6.name,
|
|
242
|
+
GROK_BUILD_0_1.name,
|
|
243
|
+
GROK_4_3.name,
|
|
244
|
+
] as const
|
|
179
245
|
|
|
180
246
|
/**
|
|
181
247
|
* Grok Image Generation Models
|
|
@@ -183,6 +249,7 @@ export const GROK_CHAT_MODELS = [GROK_BUILD_0_1.name, GROK_4_3.name] as const
|
|
|
183
249
|
export const GROK_IMAGE_MODELS = [
|
|
184
250
|
GROK_2_IMAGE.name,
|
|
185
251
|
GROK_IMAGINE_IMAGE.name,
|
|
252
|
+
GROK_IMAGINE_IMAGE_2_0.name,
|
|
186
253
|
GROK_IMAGINE_IMAGE_QUALITY.name,
|
|
187
254
|
] as const
|
|
188
255
|
|
|
@@ -228,6 +295,7 @@ const GROK_VOICE_FAST_1 = {
|
|
|
228
295
|
},
|
|
229
296
|
} as const satisfies ModelMeta
|
|
230
297
|
|
|
298
|
+
/** @deprecated xAI has deprecated grok-voice-think-fast-1.0 — use grok-voice-think-fast-2.0. */
|
|
231
299
|
const GROK_VOICE_THINK_FAST_1 = {
|
|
232
300
|
name: 'grok-voice-think-fast-1.0',
|
|
233
301
|
supports: {
|
|
@@ -238,15 +306,48 @@ const GROK_VOICE_THINK_FAST_1 = {
|
|
|
238
306
|
},
|
|
239
307
|
} as const satisfies ModelMeta
|
|
240
308
|
|
|
309
|
+
// xAI's current recommended speech-to-speech model.
|
|
310
|
+
const GROK_VOICE_THINK_FAST_2 = {
|
|
311
|
+
name: 'grok-voice-think-fast-2.0',
|
|
312
|
+
supports: {
|
|
313
|
+
input: ['audio', 'text'],
|
|
314
|
+
output: ['audio', 'text'],
|
|
315
|
+
capabilities: ['reasoning', 'tool_calling'],
|
|
316
|
+
tools: [] as const,
|
|
317
|
+
},
|
|
318
|
+
} as const satisfies ModelMeta
|
|
319
|
+
|
|
320
|
+
// Rolling alias used by xAI's realtime docs examples; always points at the
|
|
321
|
+
// latest speech-to-speech model.
|
|
322
|
+
const GROK_VOICE_LATEST = {
|
|
323
|
+
name: 'grok-voice-latest',
|
|
324
|
+
supports: {
|
|
325
|
+
input: ['audio', 'text'],
|
|
326
|
+
output: ['audio', 'text'],
|
|
327
|
+
capabilities: ['reasoning', 'tool_calling'],
|
|
328
|
+
tools: [] as const,
|
|
329
|
+
},
|
|
330
|
+
} as const satisfies ModelMeta
|
|
331
|
+
|
|
241
332
|
export const GROK_TTS_MODELS = [GROK_TTS.name] as const
|
|
242
333
|
|
|
243
334
|
export const GROK_TRANSCRIPTION_MODELS = [GROK_STT.name] as const
|
|
244
335
|
|
|
245
336
|
export const GROK_REALTIME_MODELS = [
|
|
337
|
+
GROK_VOICE_THINK_FAST_2.name,
|
|
338
|
+
GROK_VOICE_LATEST.name,
|
|
246
339
|
GROK_VOICE_FAST_1.name,
|
|
247
340
|
GROK_VOICE_THINK_FAST_1.name,
|
|
248
341
|
] as const
|
|
249
342
|
|
|
343
|
+
/**
|
|
344
|
+
* Default speech-to-speech model used by the realtime token issuer and the
|
|
345
|
+
* realtime client adapter when no model is specified. Single source of truth
|
|
346
|
+
* so a future default bump cannot leave the two sides disagreeing.
|
|
347
|
+
*/
|
|
348
|
+
export const GROK_DEFAULT_REALTIME_MODEL: GrokRealtimeModel =
|
|
349
|
+
'grok-voice-think-fast-2.0'
|
|
350
|
+
|
|
250
351
|
export type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]
|
|
251
352
|
export type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]
|
|
252
353
|
export type GrokVideoModel = (typeof GROK_VIDEO_MODELS)[number]
|
|
@@ -261,6 +362,8 @@ export type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]
|
|
|
261
362
|
export type GrokModelInputModalitiesByName = {
|
|
262
363
|
[GROK_4_3.name]: typeof GROK_4_3.supports.input
|
|
263
364
|
[GROK_BUILD_0_1.name]: typeof GROK_BUILD_0_1.supports.input
|
|
365
|
+
[GROK_4_5.name]: typeof GROK_4_5.supports.input
|
|
366
|
+
[GROK_4_6.name]: typeof GROK_4_6.supports.input
|
|
264
367
|
}
|
|
265
368
|
|
|
266
369
|
/**
|
package/src/realtime/adapter.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { resolveDebugOption } from '@tanstack/ai/adapter-internals'
|
|
2
|
+
import { GROK_DEFAULT_REALTIME_MODEL } from '../model-meta'
|
|
2
3
|
import type {
|
|
3
4
|
AnyClientTool,
|
|
4
5
|
AudioVisualization,
|
|
@@ -89,7 +90,7 @@ export function grokRealtime(
|
|
|
89
90
|
token: RealtimeToken,
|
|
90
91
|
_clientTools?: ReadonlyArray<AnyClientTool>,
|
|
91
92
|
): Promise<RealtimeConnection> {
|
|
92
|
-
const model = token.config.model ??
|
|
93
|
+
const model = token.config.model ?? GROK_DEFAULT_REALTIME_MODEL
|
|
93
94
|
logger.request(`activity=realtime provider=grok model=${model}`, {
|
|
94
95
|
provider: 'grok',
|
|
95
96
|
model,
|
|
@@ -115,7 +116,7 @@ async function createWebRTCConnection(
|
|
|
115
116
|
token: RealtimeToken,
|
|
116
117
|
logger: InternalLogger,
|
|
117
118
|
): Promise<RealtimeConnection> {
|
|
118
|
-
const model = token.config.model ??
|
|
119
|
+
const model = token.config.model ?? GROK_DEFAULT_REALTIME_MODEL
|
|
119
120
|
const eventHandlers = new Map<RealtimeEvent, Set<RealtimeEventHandler<any>>>()
|
|
120
121
|
|
|
121
122
|
const pc = new RTCPeerConnection()
|
package/src/realtime/token.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { resolveDebugOption } from '@tanstack/ai/adapter-internals'
|
|
2
|
+
import { GROK_DEFAULT_REALTIME_MODEL } from '../model-meta'
|
|
2
3
|
import { getGrokApiKeyFromEnv } from '../utils'
|
|
3
4
|
import type { RealtimeToken, RealtimeTokenAdapter } from '@tanstack/ai'
|
|
4
5
|
import type { GrokRealtimeModel } from '../model-meta'
|
|
@@ -27,7 +28,7 @@ const DEFAULT_TOKEN_FETCH_TIMEOUT_MS = 15_000
|
|
|
27
28
|
* import { grokRealtimeToken } from '@tanstack/ai-grok'
|
|
28
29
|
*
|
|
29
30
|
* const token = await realtimeToken({
|
|
30
|
-
* adapter: grokRealtimeToken({ model: 'grok-voice-fast-
|
|
31
|
+
* adapter: grokRealtimeToken({ model: 'grok-voice-think-fast-2.0' }),
|
|
31
32
|
* })
|
|
32
33
|
* ```
|
|
33
34
|
*/
|
|
@@ -41,7 +42,8 @@ export function grokRealtimeToken(
|
|
|
41
42
|
provider: 'grok',
|
|
42
43
|
|
|
43
44
|
async generateToken(): Promise<RealtimeToken> {
|
|
44
|
-
const model: GrokRealtimeModel =
|
|
45
|
+
const model: GrokRealtimeModel =
|
|
46
|
+
options.model ?? GROK_DEFAULT_REALTIME_MODEL
|
|
45
47
|
|
|
46
48
|
logger.request(`activity=realtimeToken provider=grok model=${model}`, {
|
|
47
49
|
provider: 'grok',
|
package/src/realtime/types.ts
CHANGED
|
@@ -35,7 +35,7 @@ export type GrokTurnDetection =
|
|
|
35
35
|
* Options for the Grok realtime token adapter.
|
|
36
36
|
*/
|
|
37
37
|
export interface GrokRealtimeTokenOptions {
|
|
38
|
-
/** Model to use (default: 'grok-voice-fast-
|
|
38
|
+
/** Model to use (default: 'grok-voice-think-fast-2.0'). */
|
|
39
39
|
model?: GrokRealtimeModel
|
|
40
40
|
/**
|
|
41
41
|
* Enable debug logging for token creation.
|
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Grok Video Generation Provider Options (xAI Imagine API)
|
|
3
3
|
*
|
|
4
|
-
* Based on https://docs.x.ai/
|
|
4
|
+
* Based on https://docs.x.ai/developers/model-capabilities/video/generation
|
|
5
|
+
* (plus the image-to-video, reference-to-video, editing, and extension pages
|
|
6
|
+
* under the same section).
|
|
5
7
|
*
|
|
6
8
|
* @experimental Video generation is an experimental feature and may change.
|
|
7
9
|
*/
|
|
@@ -34,6 +36,14 @@ export type GrokVideoAspectRatio =
|
|
|
34
36
|
*/
|
|
35
37
|
export type GrokVideoResolution = '480p' | '720p' | '1080p'
|
|
36
38
|
|
|
39
|
+
/**
|
|
40
|
+
* Resolutions accepted by grok-imagine-video (v1.0). Native 1080p is a
|
|
41
|
+
* grok-imagine-video-1.5 generation feature.
|
|
42
|
+
*
|
|
43
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
44
|
+
*/
|
|
45
|
+
export type GrokVideoResolutionV1 = '480p' | '720p'
|
|
46
|
+
|
|
37
47
|
/**
|
|
38
48
|
* Size strings for grok-imagine video models. The Imagine API is
|
|
39
49
|
* aspect-ratio based rather than pixel-size based; like the grok-imagine
|
|
@@ -47,6 +57,15 @@ export type GrokVideoSize =
|
|
|
47
57
|
| GrokVideoAspectRatio
|
|
48
58
|
| `${GrokVideoAspectRatio}_${GrokVideoResolution}`
|
|
49
59
|
|
|
60
|
+
/**
|
|
61
|
+
* Size strings for grok-imagine-video (v1.0) — 1080p is not in the type.
|
|
62
|
+
*
|
|
63
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
64
|
+
*/
|
|
65
|
+
export type GrokVideoSizeV1 =
|
|
66
|
+
| GrokVideoAspectRatio
|
|
67
|
+
| `${GrokVideoAspectRatio}_${GrokVideoResolutionV1}`
|
|
68
|
+
|
|
50
69
|
const GROK_VIDEO_ASPECT_RATIOS: ReadonlyArray<string> = [
|
|
51
70
|
'1:1',
|
|
52
71
|
'16:9',
|
|
@@ -80,6 +99,16 @@ export function parseGrokVideoSize(
|
|
|
80
99
|
return { aspectRatio, ...(resolution !== undefined && { resolution }) }
|
|
81
100
|
}
|
|
82
101
|
|
|
102
|
+
/**
|
|
103
|
+
* Models that accept native 1080p on text-to-video and image-to-video.
|
|
104
|
+
* Reference-to-video stays capped at 720p even on these models.
|
|
105
|
+
*
|
|
106
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
107
|
+
*/
|
|
108
|
+
export function isGrokVideoNative1080pModel(model: string): boolean {
|
|
109
|
+
return model === 'grok-imagine-video-1.5'
|
|
110
|
+
}
|
|
111
|
+
|
|
83
112
|
/**
|
|
84
113
|
* Validate the `size` template for a given grok video model.
|
|
85
114
|
*
|
|
@@ -107,6 +136,12 @@ export function validateVideoSize(
|
|
|
107
136
|
`Supported resolutions: ${GROK_VIDEO_RESOLUTIONS.join(', ')}`,
|
|
108
137
|
)
|
|
109
138
|
}
|
|
139
|
+
if (parsed.resolution === '1080p' && !isGrokVideoNative1080pModel(model)) {
|
|
140
|
+
throw new Error(
|
|
141
|
+
`Resolution "1080p" is not supported by model "${model}". ` +
|
|
142
|
+
`Use 'grok-imagine-video-1.5' for native 1080p text-to-video / image-to-video.`,
|
|
143
|
+
)
|
|
144
|
+
}
|
|
110
145
|
}
|
|
111
146
|
|
|
112
147
|
/**
|
|
@@ -162,80 +197,209 @@ export function getGrokVideoDurationOptions<TModel extends GrokVideoModel>(
|
|
|
162
197
|
}
|
|
163
198
|
|
|
164
199
|
/**
|
|
165
|
-
*
|
|
166
|
-
*
|
|
167
|
-
* `
|
|
200
|
+
* Request mode for a source-video job. `'edit'` posts to `/v1/videos/edits`
|
|
201
|
+
* (modify the source clip in place); `'extend'` posts to
|
|
202
|
+
* `/v1/videos/extensions` (continue the source clip — `duration` is the
|
|
203
|
+
* length of the **added tail**, not the total). Both require exactly one
|
|
204
|
+
* video prompt part carrying the source clip, and both are
|
|
205
|
+
* `grok-imagine-video` (v1.0) only.
|
|
206
|
+
*
|
|
207
|
+
* Output geometry (aspect ratio / resolution) is inherited from the source
|
|
208
|
+
* clip in both modes, capped at 720p, and edit outputs also inherit the
|
|
209
|
+
* source length — the adapter rejects `size`, `aspect_ratio`, `resolution`,
|
|
210
|
+
* and (in edit mode) `duration` rather than sending fields the API ignores.
|
|
168
211
|
*
|
|
169
212
|
* @experimental Video generation is an experimental feature and may change.
|
|
170
213
|
*/
|
|
171
|
-
export
|
|
214
|
+
export type GrokVideoMode = 'edit' | 'extend'
|
|
215
|
+
|
|
216
|
+
/**
|
|
217
|
+
* Provider options shared by both grok-imagine video models. These map
|
|
218
|
+
* directly onto the Imagine API request body and take precedence over the
|
|
219
|
+
* generic `size` / `duration` options when both are provided.
|
|
220
|
+
*
|
|
221
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
222
|
+
*/
|
|
223
|
+
export interface GrokVideoBaseProviderOptions {
|
|
172
224
|
/**
|
|
173
|
-
* Output aspect ratio.
|
|
225
|
+
* Output aspect ratio. Generation only — edit / extend outputs inherit
|
|
226
|
+
* the source clip's geometry and the adapter rejects this in those modes.
|
|
174
227
|
*/
|
|
175
228
|
aspect_ratio?: GrokVideoAspectRatio
|
|
176
229
|
|
|
177
230
|
/**
|
|
178
|
-
* Output resolution tier.
|
|
231
|
+
* Output resolution tier. Generation only — edit / extend outputs inherit
|
|
232
|
+
* the source clip's geometry and the adapter rejects this in those modes.
|
|
233
|
+
* `1080p` is grok-imagine-video-1.5 generation only; reference-to-video
|
|
234
|
+
* is capped at 720p.
|
|
179
235
|
*/
|
|
180
236
|
resolution?: GrokVideoResolution
|
|
181
237
|
|
|
182
238
|
/**
|
|
183
|
-
* Video duration in integer seconds (1–15).
|
|
239
|
+
* Video duration in integer seconds (1–15). In `'extend'` mode this is
|
|
240
|
+
* the length of the added tail only, not the total output length. Not
|
|
241
|
+
* valid in `'edit'` mode (the output inherits the source clip's length).
|
|
184
242
|
*/
|
|
185
243
|
duration?: number
|
|
186
244
|
}
|
|
187
245
|
|
|
188
246
|
/**
|
|
189
|
-
*
|
|
247
|
+
* Provider options for grok-imagine-video (v1.0), which is the only model
|
|
248
|
+
* that accepts a source-video edit / extend job.
|
|
190
249
|
*
|
|
191
250
|
* @experimental Video generation is an experimental feature and may change.
|
|
192
251
|
*/
|
|
193
|
-
export
|
|
194
|
-
|
|
195
|
-
|
|
252
|
+
export interface GrokVideoSourceProviderOptions extends GrokVideoBaseProviderOptions {
|
|
253
|
+
/**
|
|
254
|
+
* Selects the request mode for a source-video prompt part: `'edit'`
|
|
255
|
+
* (`/v1/videos/edits`) or `'extend'` (`/v1/videos/extensions`). Required
|
|
256
|
+
* when the prompt carries a video part; not valid without one. Omit for
|
|
257
|
+
* plain generation (`/v1/videos/generations`). grok-imagine-video only.
|
|
258
|
+
*/
|
|
259
|
+
mode?: GrokVideoMode
|
|
196
260
|
}
|
|
197
261
|
|
|
198
262
|
/**
|
|
199
|
-
*
|
|
263
|
+
* Provider options for grok-imagine-video-1.5, which adds the
|
|
264
|
+
* reference-to-video inputs on top of the shared options.
|
|
200
265
|
*
|
|
201
266
|
* @experimental Video generation is an experimental feature and may change.
|
|
202
267
|
*/
|
|
203
|
-
export
|
|
204
|
-
|
|
205
|
-
|
|
268
|
+
export interface GrokVideoProviderOptions extends GrokVideoBaseProviderOptions {
|
|
269
|
+
/**
|
|
270
|
+
* Reference images for reference-to-video generation (output capped at
|
|
271
|
+
* 720p). Usually populated from image prompt parts with
|
|
272
|
+
* `metadata.role: 'reference'` (or `'character'`); set explicitly to
|
|
273
|
+
* replace the part-derived list. Reference images are addressed from the
|
|
274
|
+
* prompt text as `<IMAGE_0>`, `<IMAGE_1>`, … in request order, and do not
|
|
275
|
+
* lock the first frame.
|
|
276
|
+
*/
|
|
277
|
+
reference_images?: Array<{ url: string }>
|
|
278
|
+
|
|
279
|
+
/**
|
|
280
|
+
* Preset TTS voices to reference for generated speech (max 3). Voice ids
|
|
281
|
+
* come from the xAI TTS voice roster (e.g. 'eve', 'rex') or a custom
|
|
282
|
+
* voice id, and are addressed from the prompt text as `<AUDIO_0>`,
|
|
283
|
+
* `<AUDIO_1>`, `<AUDIO_2>`.
|
|
284
|
+
*/
|
|
285
|
+
reference_audios?: Array<{ voice_id: string }>
|
|
206
286
|
}
|
|
207
287
|
|
|
208
288
|
/**
|
|
209
|
-
*
|
|
210
|
-
*
|
|
211
|
-
*
|
|
212
|
-
* `grok-imagine-video-1.5` is image-to-video only (the image is required).
|
|
289
|
+
* Widest option surface. Used when `modelOptions` arrives as deserialized
|
|
290
|
+
* JSON and the adapter must validate fields the per-model map already
|
|
291
|
+
* hides at compile time.
|
|
213
292
|
*
|
|
214
293
|
* @experimental Video generation is an experimental feature and may change.
|
|
215
294
|
*/
|
|
216
|
-
export type
|
|
217
|
-
|
|
218
|
-
|
|
295
|
+
export type GrokVideoRuntimeOptions = GrokVideoSourceProviderOptions &
|
|
296
|
+
GrokVideoProviderOptions
|
|
297
|
+
|
|
298
|
+
/**
|
|
299
|
+
* Maximum reference voices accepted by the Imagine video endpoint.
|
|
300
|
+
*/
|
|
301
|
+
export const GROK_VIDEO_MAX_REFERENCE_AUDIOS = 3
|
|
302
|
+
|
|
303
|
+
/**
|
|
304
|
+
* Maximum reference images accepted by the Imagine video endpoint.
|
|
305
|
+
*/
|
|
306
|
+
export const GROK_VIDEO_MAX_REFERENCE_IMAGES = 7
|
|
307
|
+
|
|
308
|
+
/**
|
|
309
|
+
* Model names whose per-model options declare the reference fields. Keeps
|
|
310
|
+
* the runtime set below provably in sync with
|
|
311
|
+
* {@link GrokVideoModelProviderOptionsByName} — a typo or a new
|
|
312
|
+
* reference-capable model missing from the set is a compile error.
|
|
313
|
+
*/
|
|
314
|
+
type GrokVideoReferenceModel = {
|
|
315
|
+
[TModel in GrokVideoModel]: 'reference_images' extends keyof GrokVideoModelProviderOptionsByName[TModel]
|
|
316
|
+
? TModel
|
|
317
|
+
: never
|
|
318
|
+
}[GrokVideoModel]
|
|
319
|
+
|
|
320
|
+
/**
|
|
321
|
+
* Models that support reference-to-video inputs (`reference_images` /
|
|
322
|
+
* `reference_audios`). The per-model provider-options map hides the fields
|
|
323
|
+
* from other models at compile time; this backs the runtime gate for
|
|
324
|
+
* untyped callers (e.g. deserialized JSON) so they get a clear error
|
|
325
|
+
* instead of a raw API 400.
|
|
326
|
+
*
|
|
327
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
328
|
+
*/
|
|
329
|
+
const GROK_VIDEO_REFERENCE_MODELS: ReadonlySet<string> =
|
|
330
|
+
new Set<GrokVideoReferenceModel>(['grok-imagine-video-1.5'])
|
|
331
|
+
|
|
332
|
+
/**
|
|
333
|
+
* True when the model accepts reference-to-video inputs.
|
|
334
|
+
*
|
|
335
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
336
|
+
*/
|
|
337
|
+
export function isGrokVideoReferenceModel(model: string): boolean {
|
|
338
|
+
return GROK_VIDEO_REFERENCE_MODELS.has(model)
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
/**
|
|
342
|
+
* Model names whose per-model options declare `mode`. Same
|
|
343
|
+
* provably-in-sync construction as {@link GrokVideoReferenceModel}.
|
|
344
|
+
*/
|
|
345
|
+
type GrokVideoSourceModel = {
|
|
346
|
+
[TModel in GrokVideoModel]: 'mode' extends keyof GrokVideoModelProviderOptionsByName[TModel]
|
|
347
|
+
? TModel
|
|
348
|
+
: never
|
|
349
|
+
}[GrokVideoModel]
|
|
350
|
+
|
|
351
|
+
/**
|
|
352
|
+
* Models that accept a source-video prompt part for `/v1/videos/edits`
|
|
353
|
+
* and `/v1/videos/extensions`. xAI lists video input only on
|
|
354
|
+
* grok-imagine-video (v1.0).
|
|
355
|
+
*
|
|
356
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
357
|
+
*/
|
|
358
|
+
const GROK_VIDEO_SOURCE_MODELS: ReadonlySet<string> =
|
|
359
|
+
new Set<GrokVideoSourceModel>(['grok-imagine-video'])
|
|
360
|
+
|
|
361
|
+
/**
|
|
362
|
+
* True when the model accepts edit / extend source-video jobs.
|
|
363
|
+
*
|
|
364
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
365
|
+
*/
|
|
366
|
+
export function isGrokVideoSourceModel(model: string): boolean {
|
|
367
|
+
return GROK_VIDEO_SOURCE_MODELS.has(model)
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
/**
|
|
371
|
+
* Type-only map from model name to its specific provider options. Only
|
|
372
|
+
* grok-imagine-video-1.5 exposes the reference-to-video fields. Only
|
|
373
|
+
* grok-imagine-video (v1.0) exposes `mode` for edit / extend.
|
|
374
|
+
*
|
|
375
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
376
|
+
*/
|
|
377
|
+
export type GrokVideoModelProviderOptionsByName = {
|
|
378
|
+
'grok-imagine-video': GrokVideoSourceProviderOptions
|
|
379
|
+
'grok-imagine-video-1.5': GrokVideoProviderOptions
|
|
219
380
|
}
|
|
220
381
|
|
|
221
382
|
/**
|
|
222
|
-
*
|
|
223
|
-
* required and text-to-video is rejected by the Imagine API. Used by the
|
|
224
|
-
* adapter to fail fast with a clear message instead of surfacing the raw
|
|
225
|
-
* "Text-to-video is not supported for this model" 400.
|
|
383
|
+
* Type-only map from model name to its supported `size` strings.
|
|
226
384
|
*
|
|
227
385
|
* @experimental Video generation is an experimental feature and may change.
|
|
228
386
|
*/
|
|
229
|
-
|
|
230
|
-
'grok-imagine-video
|
|
231
|
-
|
|
387
|
+
export type GrokVideoModelSizeByName = {
|
|
388
|
+
'grok-imagine-video': GrokVideoSizeV1
|
|
389
|
+
'grok-imagine-video-1.5': GrokVideoSize
|
|
390
|
+
}
|
|
232
391
|
|
|
233
392
|
/**
|
|
234
|
-
*
|
|
235
|
-
*
|
|
393
|
+
* Type-only map from model name to the non-text prompt modalities it accepts.
|
|
394
|
+
* Both models support text-to-video and accept an optional `image` prompt
|
|
395
|
+
* part as the starting frame; image parts with `metadata.role: 'reference'`
|
|
396
|
+
* or `'character'` become `reference_images` (grok-imagine-video-1.5 only).
|
|
397
|
+
* A `video` prompt part carries the source clip for edit / extension mode
|
|
398
|
+
* on grok-imagine-video only (`modelOptions.mode: 'edit' | 'extend'`).
|
|
236
399
|
*
|
|
237
400
|
* @experimental Video generation is an experimental feature and may change.
|
|
238
401
|
*/
|
|
239
|
-
export
|
|
240
|
-
|
|
402
|
+
export type GrokVideoModelInputModalitiesByName = {
|
|
403
|
+
'grok-imagine-video': readonly ['image', 'video']
|
|
404
|
+
'grok-imagine-video-1.5': readonly ['image']
|
|
241
405
|
}
|