@tanstack/ai-grok 0.13.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,462 @@
1
+ import { resolveMediaPrompt } from '@tanstack/ai'
2
+ import { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'
3
+ import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
4
+ import { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'
5
+ import {
6
+ getGrokVideoDurationOptions,
7
+ isImageToVideoOnlyModel,
8
+ parseGrokVideoSize,
9
+ validateVideoSize,
10
+ } from '../video/video-provider-options'
11
+ import type { DurationOptions } from '@tanstack/ai/adapters'
12
+ import type {
13
+ ImagePart,
14
+ MediaInputMetadata,
15
+ TokenUsage,
16
+ VideoGenerationOptions,
17
+ VideoJobResult,
18
+ VideoStatusResult,
19
+ VideoUrlResult,
20
+ } from '@tanstack/ai'
21
+ import type { GrokVideoModel } from '../model-meta'
22
+ import type {
23
+ GrokVideoModelDurationByName,
24
+ GrokVideoModelInputModalitiesByName,
25
+ GrokVideoModelProviderOptionsByName,
26
+ GrokVideoModelSizeByName,
27
+ GrokVideoProviderOptions,
28
+ } from '../video/video-provider-options'
29
+ import type { GrokClientConfig } from '../utils'
30
+
31
+ /**
32
+ * Configuration for Grok video adapter.
33
+ *
34
+ * @experimental Video generation is an experimental feature and may change.
35
+ */
36
+ export interface GrokVideoConfig extends GrokClientConfig {}
37
+
38
+ /**
39
+ * xAI bills video generation in "USD ticks": 10^10 ticks per US dollar
40
+ * (e.g. one grok-imagine-video-1.5 second costs $0.08 = 800_000_000 ticks).
41
+ */
42
+ const USD_TICKS_PER_DOLLAR = 10_000_000_000
43
+
44
+ /** Response of POST /v1/videos/generations. */
45
+ interface GrokVideoCreateResponse {
46
+ request_id?: string
47
+ }
48
+
49
+ /** Response of GET /v1/videos/{request_id}. */
50
+ interface GrokVideoStatusResponse {
51
+ status?: string
52
+ progress?: number
53
+ model?: string
54
+ video?: {
55
+ url?: string
56
+ duration?: number
57
+ }
58
+ usage?: {
59
+ cost_in_usd_ticks?: number
60
+ }
61
+ error?: string
62
+ }
63
+
64
+ /**
65
+ * Convert a TanStack ImagePart to the URL string accepted by xAI's Imagine
66
+ * video endpoint: public URLs pass through (fetched by xAI's servers), data
67
+ * sources become base64 data URIs.
68
+ */
69
+ function imagePartToUrl(part: ImagePart<MediaInputMetadata>): string {
70
+ if (part.source.type === 'url') return part.source.value
71
+ return `data:${part.source.mimeType};base64,${part.source.value}`
72
+ }
73
+
74
+ function buildGrokVideoUsage(
75
+ response: GrokVideoStatusResponse,
76
+ ): TokenUsage | undefined {
77
+ const seconds = response.video?.duration
78
+ const ticks = response.usage?.cost_in_usd_ticks
79
+ if (seconds === undefined && ticks === undefined) return undefined
80
+ return {
81
+ promptTokens: 0,
82
+ completionTokens: 0,
83
+ totalTokens: 0,
84
+ ...(seconds !== undefined && { unitsBilled: seconds }),
85
+ ...(ticks !== undefined && { cost: ticks / USD_TICKS_PER_DOLLAR }),
86
+ }
87
+ }
88
+
89
+ /**
90
+ * Grok Video Generation Adapter (xAI Imagine API)
91
+ *
92
+ * Tree-shakeable adapter for the grok-imagine video models using the
93
+ * async jobs/polling architecture: create a generation request, poll it,
94
+ * then read the completed video URL.
95
+ *
96
+ * `grok-imagine-video` (v1.0) supports text-to-video and image-to-video.
97
+ * `grok-imagine-video-1.5` is image-to-video only — every request needs an
98
+ * image prompt part as the starting frame, and the adapter rejects a
99
+ * text-only prompt with a clear error rather than a raw API 400.
100
+ *
101
+ * The Imagine video endpoints are not part of the OpenAI SDK surface (and
102
+ * xAI rejects the SDK's multipart paths), so requests are plain JSON calls
103
+ * issued with the configured `fetch` (or the global one).
104
+ *
105
+ * @experimental Video generation is an experimental feature and may change.
106
+ *
107
+ * Features:
108
+ * - Async job-based video generation (1–15 second clips with audio)
109
+ * - Aspect-ratio sizing via the "aspectRatio_resolution" size template
110
+ * (e.g. '16:9_720p'), consistent with the grok-imagine image models
111
+ * - Image-to-video via an `image` prompt part (starting frame URL or data URI)
112
+ * - Usage reporting: billed seconds (`unitsBilled`) and exact cost
113
+ */
114
+ export class GrokVideoAdapter<
115
+ TModel extends GrokVideoModel,
116
+ > extends BaseVideoAdapter<
117
+ TModel,
118
+ GrokVideoProviderOptions,
119
+ GrokVideoModelProviderOptionsByName,
120
+ GrokVideoModelSizeByName,
121
+ GrokVideoModelInputModalitiesByName,
122
+ GrokVideoModelDurationByName
123
+ > {
124
+ readonly name = 'grok' as const
125
+
126
+ private readonly clientConfig: GrokVideoConfig
127
+
128
+ constructor(config: GrokVideoConfig, model: TModel) {
129
+ super({}, model)
130
+ this.clientConfig = withGrokDefaults(config)
131
+ }
132
+
133
+ private get fetch(): (
134
+ input: string,
135
+ init?: RequestInit,
136
+ ) => Promise<Response> {
137
+ return this.clientConfig.fetch ?? fetch
138
+ }
139
+
140
+ private async request(
141
+ path: string,
142
+ init?: Omit<RequestInit, 'headers'>,
143
+ ): Promise<Response> {
144
+ return await this.fetch(`${this.clientConfig.baseURL}${path}`, {
145
+ ...init,
146
+ headers: {
147
+ 'Content-Type': 'application/json',
148
+ Authorization: `Bearer ${this.clientConfig.apiKey}`,
149
+ },
150
+ })
151
+ }
152
+
153
+ /**
154
+ * Reads the error message out of an Imagine API error body
155
+ * (`{"code": "...", "error": "..."}`), falling back to the raw text.
156
+ */
157
+ private async errorMessage(response: Response): Promise<string> {
158
+ const body = await response.text()
159
+ try {
160
+ const parsed: unknown = JSON.parse(body)
161
+ if (
162
+ typeof parsed === 'object' &&
163
+ parsed !== null &&
164
+ 'error' in parsed &&
165
+ typeof parsed.error === 'string'
166
+ ) {
167
+ return parsed.error
168
+ }
169
+ } catch {
170
+ // not JSON — fall through to the raw body
171
+ }
172
+ return body
173
+ }
174
+
175
+ async createVideoJob(
176
+ options: VideoGenerationOptions<
177
+ GrokVideoProviderOptions,
178
+ GrokVideoModelSizeByName[TModel],
179
+ GrokVideoModelDurationByName[TModel]
180
+ >,
181
+ ): Promise<VideoJobResult> {
182
+ const { model, size, modelOptions, logger } = options
183
+
184
+ validateVideoSize(model, size)
185
+
186
+ // Coerce the requested duration into the model's valid range (1–15s,
187
+ // integer) instead of rejecting it — `snapDuration` clamps and rounds.
188
+ // modelOptions wins over the generic `duration`, mirroring the size
189
+ // precedence below.
190
+ const rawDuration = modelOptions?.duration ?? options.duration
191
+ const duration =
192
+ rawDuration !== undefined ? this.snapDuration(rawDuration) : undefined
193
+
194
+ // The interleaved prompt decomposes into verbatim text plus typed media
195
+ // buckets. The Imagine video endpoint takes a text prompt and an optional
196
+ // starting frame; reject the modalities it can't consume.
197
+ const resolved = resolveMediaPrompt(options.prompt)
198
+ if (resolved.videos.length > 0) {
199
+ throw new Error(
200
+ `${this.name}.createVideoJob does not support video prompt parts (model: ${model}).`,
201
+ )
202
+ }
203
+ if (resolved.audios.length > 0) {
204
+ throw new Error(
205
+ `${this.name}.createVideoJob does not support audio prompt parts (model: ${model}).`,
206
+ )
207
+ }
208
+ // grok-imagine-video-1.5 is image-to-video only — text-to-video is
209
+ // rejected by the API, so fail fast with a clear, actionable message
210
+ // pointing at the model that does support text-to-video.
211
+ if (resolved.images.length === 0 && isImageToVideoOnlyModel(model)) {
212
+ throw new Error(
213
+ `${this.name}: ${model} does not support text-to-video — it is image-to-video only. ` +
214
+ `Include an image prompt part as the starting frame, or use 'grok-imagine-video' for text-to-video.`,
215
+ )
216
+ }
217
+ if (resolved.images.length > 1) {
218
+ throw new Error(
219
+ `${this.name}: ${model} accepts at most one starting-frame image; received ${resolved.images.length}.`,
220
+ )
221
+ }
222
+
223
+ // Image-to-video: the single image prompt part becomes the starting frame
224
+ // and the prompt text describes the desired motion. URL sources are
225
+ // fetched by xAI's servers; data sources are sent as base64 data URIs.
226
+ const [startFrame] = resolved.images
227
+
228
+ // The generic `size` option carries an "aspectRatio_resolution" template
229
+ // (e.g. '16:9_720p') and maps to the Imagine API's `aspect_ratio` /
230
+ // `resolution` parameters; explicit modelOptions win over the template.
231
+ const parsedSize = size !== undefined ? parseGrokVideoSize(size) : undefined
232
+ const request = {
233
+ model,
234
+ prompt: resolved.text,
235
+ ...(startFrame && { image: { url: imagePartToUrl(startFrame) } }),
236
+ ...(parsedSize && {
237
+ aspect_ratio: parsedSize.aspectRatio,
238
+ ...(parsedSize.resolution !== undefined && {
239
+ resolution: parsedSize.resolution,
240
+ }),
241
+ }),
242
+ ...modelOptions,
243
+ // Spread after modelOptions so the snapped duration is authoritative
244
+ // (modelOptions.duration is folded into `duration` via snapDuration above).
245
+ ...(duration !== undefined && { duration }),
246
+ }
247
+
248
+ try {
249
+ logger.request(
250
+ `activity=video.create provider=${this.name} model=${model} size=${size ?? 'default'} duration=${duration ?? 'default'}`,
251
+ { provider: this.name, model },
252
+ )
253
+
254
+ const response = await this.request('/videos/generations', {
255
+ method: 'POST',
256
+ body: JSON.stringify(request),
257
+ })
258
+ if (!response.ok) {
259
+ throw new Error(
260
+ `grok: video generation request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,
261
+ )
262
+ }
263
+
264
+ const result = (await response.json()) as GrokVideoCreateResponse
265
+ if (!result.request_id) {
266
+ throw new Error(
267
+ 'grok: video generation response contained no request_id',
268
+ )
269
+ }
270
+ return { jobId: result.request_id, model }
271
+ } catch (error: unknown) {
272
+ logger.errors(`${this.name}.createVideoJob fatal`, {
273
+ error: toRunErrorPayload(error, `${this.name}.createVideoJob failed`),
274
+ source: `${this.name}.createVideoJob`,
275
+ })
276
+ throw error
277
+ }
278
+ }
279
+
280
+ private async retrieveJob(jobId: string): Promise<GrokVideoStatusResponse> {
281
+ const response = await this.request(`/videos/${jobId}`)
282
+ if (!response.ok) {
283
+ const error = new Error(
284
+ `grok: video status request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,
285
+ )
286
+ ;(error as { status?: number }).status = response.status
287
+ throw error
288
+ }
289
+ return (await response.json()) as GrokVideoStatusResponse
290
+ }
291
+
292
+ async getVideoStatus(jobId: string): Promise<VideoStatusResult> {
293
+ let response: GrokVideoStatusResponse
294
+ try {
295
+ response = await this.retrieveJob(jobId)
296
+ } catch (error) {
297
+ if ((error as { status?: number }).status === 404) {
298
+ return { jobId, status: 'failed', error: 'Job not found' }
299
+ }
300
+ throw error
301
+ }
302
+
303
+ return {
304
+ jobId,
305
+ status: this.mapStatus(response.status),
306
+ ...(response.progress !== undefined && { progress: response.progress }),
307
+ ...(response.error !== undefined && { error: response.error }),
308
+ }
309
+ }
310
+
311
+ async getVideoUrl(jobId: string): Promise<VideoUrlResult> {
312
+ let response: GrokVideoStatusResponse
313
+ try {
314
+ response = await this.retrieveJob(jobId)
315
+ } catch (error) {
316
+ if ((error as { status?: number }).status === 404) {
317
+ throw new Error(`Video job not found: ${jobId}`)
318
+ }
319
+ throw error
320
+ }
321
+
322
+ const status = this.mapStatus(response.status)
323
+ if (status === 'failed') {
324
+ throw new Error(
325
+ `Video generation failed${response.error ? `: ${response.error}` : ''}. Job ID: ${jobId}`,
326
+ )
327
+ }
328
+ const url = response.video?.url
329
+ if (!url) {
330
+ throw new Error(
331
+ `Video is not ready for download. Check status first. Job ID: ${jobId}`,
332
+ )
333
+ }
334
+
335
+ const usage = buildGrokVideoUsage(response)
336
+ return {
337
+ jobId,
338
+ url,
339
+ ...(usage && { usage }),
340
+ }
341
+ }
342
+
343
+ /**
344
+ * Maps Imagine API job statuses onto the generic video status set. The
345
+ * API reports 'pending' while queued/generating (with a numeric
346
+ * `progress`), then a terminal 'done' / 'failed' / 'expired'.
347
+ */
348
+ protected mapStatus(
349
+ apiStatus: string | undefined,
350
+ ): 'pending' | 'processing' | 'completed' | 'failed' {
351
+ switch (apiStatus) {
352
+ case 'pending':
353
+ case 'queued':
354
+ return 'pending'
355
+ case 'done':
356
+ case 'completed':
357
+ case 'succeeded':
358
+ return 'completed'
359
+ case 'failed':
360
+ case 'expired':
361
+ case 'error':
362
+ case 'cancelled':
363
+ return 'failed'
364
+ case undefined:
365
+ default:
366
+ return 'processing'
367
+ }
368
+ }
369
+
370
+ /**
371
+ * Both grok-imagine video models accept a continuous 1–15 integer-second
372
+ * range. Consumers can use this to render UI without provider knowledge.
373
+ */
374
+ override availableDurations(): DurationOptions<
375
+ GrokVideoModelDurationByName[TModel]
376
+ > {
377
+ return getGrokVideoDurationOptions(this.model)
378
+ }
379
+
380
+ /**
381
+ * Coerce a raw seconds value to the closest valid duration (clamped to
382
+ * [1, 15] and rounded to whole seconds).
383
+ */
384
+ override snapDuration(
385
+ seconds: number,
386
+ ): GrokVideoModelDurationByName[TModel] | undefined {
387
+ return snapToDurationOption(seconds, this.availableDurations())
388
+ }
389
+ }
390
+
391
+ /**
392
+ * Creates a Grok video adapter with an explicit API key.
393
+ * Type resolution happens here at the call site.
394
+ *
395
+ * @experimental Video generation is an experimental feature and may change.
396
+ *
397
+ * @param model - The model name (e.g., 'grok-imagine-video')
398
+ * @param apiKey - Your xAI API key
399
+ * @param config - Optional additional configuration
400
+ * @returns Configured Grok video adapter instance with resolved types
401
+ *
402
+ * @example
403
+ * ```typescript
404
+ * // grok-imagine-video (v1.0) supports text-to-video.
405
+ * const adapter = createGrokVideo('grok-imagine-video', 'xai-...');
406
+ *
407
+ * const { jobId } = await generateVideo({
408
+ * adapter,
409
+ * prompt: 'A beautiful sunset over the ocean',
410
+ * size: '16:9_720p',
411
+ * duration: 5
412
+ * });
413
+ * ```
414
+ */
415
+ export function createGrokVideo<TModel extends GrokVideoModel>(
416
+ model: TModel,
417
+ apiKey: string,
418
+ config?: Omit<GrokVideoConfig, 'apiKey'>,
419
+ ): GrokVideoAdapter<TModel> {
420
+ return new GrokVideoAdapter({ apiKey, ...config }, model)
421
+ }
422
+
423
+ /**
424
+ * Creates a Grok video adapter with automatic API key detection from environment variables.
425
+ * Type resolution happens here at the call site.
426
+ *
427
+ * Looks for `XAI_API_KEY` in:
428
+ * - `process.env` (Node.js)
429
+ * - `window.env` (Browser with injected env)
430
+ *
431
+ * @experimental Video generation is an experimental feature and may change.
432
+ *
433
+ * @param model - The model name (e.g., 'grok-imagine-video-1.5')
434
+ * @param config - Optional configuration (excluding apiKey which is auto-detected)
435
+ * @returns Configured Grok video adapter instance with resolved types
436
+ * @throws Error if XAI_API_KEY is not found in environment
437
+ *
438
+ * @example
439
+ * ```typescript
440
+ * // Automatically uses XAI_API_KEY from environment
441
+ * const adapter = grokVideo('grok-imagine-video-1.5');
442
+ *
443
+ * // Image-to-video only: the prompt must carry a starting-frame image part.
444
+ * const { jobId } = await generateVideo({
445
+ * adapter,
446
+ * prompt: [
447
+ * { type: 'text', content: 'Make the cat start playing the piano' },
448
+ * { type: 'image', source: { type: 'url', value: 'https://example.com/cat.png' } },
449
+ * ],
450
+ * });
451
+ *
452
+ * // Poll for status
453
+ * const status = await getVideoJobStatus({ adapter, jobId });
454
+ * ```
455
+ */
456
+ export function grokVideo<TModel extends GrokVideoModel>(
457
+ model: TModel,
458
+ config?: Omit<GrokVideoConfig, 'apiKey'>,
459
+ ): GrokVideoAdapter<TModel> {
460
+ const apiKey = getGrokApiKeyFromEnv()
461
+ return createGrokVideo(model, apiKey, config)
462
+ }
package/src/index.ts CHANGED
@@ -31,6 +31,27 @@ export type {
31
31
  GrokImageModelProviderOptionsByName,
32
32
  } from './image/image-provider-options'
33
33
 
34
+ // Video adapter - for video generation (xAI Imagine API)
35
+ export {
36
+ GrokVideoAdapter,
37
+ createGrokVideo,
38
+ grokVideo,
39
+ type GrokVideoConfig,
40
+ } from './adapters/video'
41
+ export {
42
+ GROK_VIDEO_DURATIONS,
43
+ getGrokVideoDurationOptions,
44
+ } from './video/video-provider-options'
45
+ export type {
46
+ GrokVideoProviderOptions,
47
+ GrokVideoModelProviderOptionsByName,
48
+ GrokVideoModelSizeByName,
49
+ GrokVideoModelDurationByName,
50
+ GrokVideoAspectRatio,
51
+ GrokVideoResolution,
52
+ GrokVideoSize,
53
+ } from './video/video-provider-options'
54
+
34
55
  // Speech (TTS) adapter - for text-to-speech
35
56
  export {
36
57
  GrokSpeechAdapter,
@@ -68,6 +89,7 @@ export type {
68
89
  ResolveInputModalities,
69
90
  GrokChatModel,
70
91
  GrokImageModel,
92
+ GrokVideoModel,
71
93
  GrokTTSModel,
72
94
  GrokTranscriptionModel,
73
95
  GrokRealtimeModel,
@@ -75,6 +97,7 @@ export type {
75
97
  export {
76
98
  GROK_CHAT_MODELS,
77
99
  GROK_IMAGE_MODELS,
100
+ GROK_VIDEO_MODELS,
78
101
  GROK_TTS_MODELS,
79
102
  GROK_TRANSCRIPTION_MODELS,
80
103
  GROK_REALTIME_MODELS,
package/src/model-meta.ts CHANGED
@@ -91,6 +91,47 @@ const GROK_IMAGINE_IMAGE_QUALITY = {
91
91
  },
92
92
  } as const satisfies ModelMeta
93
93
 
94
+ // Imagine API video models. Pricing is per second of generated video
95
+ // (output only); generated videos carry an audio track.
96
+ //
97
+ // grok-imagine-video (v1.0) supports both text-to-video (a starting image is
98
+ // optional) and image-to-video. grok-imagine-video-1.5 is image-to-video
99
+ // only: a starting-frame image is required (the text prompt describes the
100
+ // desired motion) — its text-to-video is rejected by the API.
101
+ const GROK_IMAGINE_VIDEO = {
102
+ name: 'grok-imagine-video',
103
+ supports: {
104
+ input: ['text', 'image'],
105
+ output: ['video', 'audio'],
106
+ },
107
+ pricing: {
108
+ input: {
109
+ normal: 0,
110
+ },
111
+ output: {
112
+ // per second of video
113
+ normal: 0.05,
114
+ },
115
+ },
116
+ } as const satisfies ModelMeta
117
+
118
+ const GROK_IMAGINE_VIDEO_1_5 = {
119
+ name: 'grok-imagine-video-1.5',
120
+ supports: {
121
+ input: ['text', 'image'],
122
+ output: ['video', 'audio'],
123
+ },
124
+ pricing: {
125
+ input: {
126
+ normal: 0,
127
+ },
128
+ output: {
129
+ // per second of video
130
+ normal: 0.08,
131
+ },
132
+ },
133
+ } as const satisfies ModelMeta
134
+
94
135
  const GROK_4_3 = {
95
136
  name: 'grok-4.3',
96
137
  context_window: 1_000_000,
@@ -145,6 +186,16 @@ export const GROK_IMAGE_MODELS = [
145
186
  GROK_IMAGINE_IMAGE_QUALITY.name,
146
187
  ] as const
147
188
 
189
+ /**
190
+ * Grok Video Generation Models (xAI Imagine API)
191
+ *
192
+ * @experimental Video generation is an experimental feature and may change.
193
+ */
194
+ export const GROK_VIDEO_MODELS = [
195
+ GROK_IMAGINE_VIDEO.name,
196
+ GROK_IMAGINE_VIDEO_1_5.name,
197
+ ] as const
198
+
148
199
  // xAI's `/v1/tts` endpoint is endpoint-addressed and does not take a `model`
149
200
  // parameter. This synthetic identifier satisfies the SDK's `TTSOptions.model`
150
201
  // contract and provides a stable value for logging and fixture matching.
@@ -198,6 +249,7 @@ export const GROK_REALTIME_MODELS = [
198
249
 
199
250
  export type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]
200
251
  export type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]
252
+ export type GrokVideoModel = (typeof GROK_VIDEO_MODELS)[number]
201
253
  export type GrokTTSModel = (typeof GROK_TTS_MODELS)[number]
202
254
  export type GrokTranscriptionModel = (typeof GROK_TRANSCRIPTION_MODELS)[number]
203
255
  export type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]