@tanstack/ai-grok 0.12.4 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/model-meta.ts CHANGED
@@ -1,13 +1,18 @@
1
1
  /**
2
2
  * Model metadata interface for documentation and type inference
3
3
  */
4
+ import type {
5
+ GrokBuildProviderOptions,
6
+ GrokTextProviderOptions,
7
+ } from './text/text-provider-options'
8
+
4
9
  interface ModelMeta {
5
10
  name: string
6
11
  supports: {
7
12
  input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>
8
13
  output: Array<'text' | 'image' | 'audio' | 'video'>
9
14
  capabilities?: Array<'reasoning' | 'tool_calling' | 'structured_outputs'>
10
- tools?: ReadonlyArray<never>
15
+ tools?: ReadonlyArray<GrokProviderToolKind>
11
16
  }
12
17
  max_input_tokens?: number
13
18
  max_output_tokens?: number
@@ -24,184 +29,18 @@ interface ModelMeta {
24
29
  }
25
30
  }
26
31
 
27
- const GROK_4_1_FAST_REASONING = {
28
- name: 'grok-4-1-fast-reasoning',
29
- context_window: 2_000_000,
30
- supports: {
31
- input: ['text', 'image'],
32
- output: ['text'],
33
- capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
34
- tools: [] as const,
35
- },
36
- pricing: {
37
- input: {
38
- normal: 0.2,
39
- cached: 0.05,
40
- },
41
- output: {
42
- normal: 0.5,
43
- },
44
- },
45
- } as const satisfies ModelMeta
32
+ export type GrokProviderToolKind =
33
+ | 'web_search'
34
+ | 'x_search'
35
+ | 'file_search'
36
+ | 'mcp'
46
37
 
47
- const GROK_4_1_FAST_NON_REASONING = {
48
- name: 'grok-4-1-fast-non-reasoning',
49
- context_window: 2_000_000,
50
- supports: {
51
- input: ['text', 'image'],
52
- output: ['text'],
53
- capabilities: ['structured_outputs', 'tool_calling'],
54
- tools: [] as const,
55
- },
56
- pricing: {
57
- input: {
58
- normal: 0.2,
59
- cached: 0.05,
60
- },
61
- output: {
62
- normal: 0.5,
63
- },
64
- },
65
- } as const satisfies ModelMeta
66
-
67
- const GROK_CODE_FAST_1 = {
68
- name: 'grok-code-fast-1',
69
- context_window: 256_000,
70
- supports: {
71
- input: ['text'],
72
- output: ['text'],
73
- capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
74
- tools: [] as const,
75
- },
76
- pricing: {
77
- input: {
78
- normal: 0.2,
79
- cached: 0.02,
80
- },
81
- output: {
82
- normal: 1.5,
83
- },
84
- },
85
- } as const satisfies ModelMeta
86
-
87
- const GROK_4_FAST_REASONING = {
88
- name: 'grok-4-fast-reasoning',
89
- context_window: 2_000_000,
90
- supports: {
91
- input: ['text', 'image'],
92
- output: ['text'],
93
- capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
94
- tools: [] as const,
95
- },
96
- pricing: {
97
- input: {
98
- normal: 0.2,
99
- cached: 0.05,
100
- },
101
- output: {
102
- normal: 0.5,
103
- },
104
- },
105
- } as const satisfies ModelMeta
106
-
107
- const GROK_4_FAST_NON_REASONING = {
108
- name: 'grok-4-fast-non-reasoning',
109
- context_window: 2_000_000,
110
- supports: {
111
- input: ['text', 'image'],
112
- output: ['text'],
113
- capabilities: ['structured_outputs', 'tool_calling'],
114
- tools: [] as const,
115
- },
116
- pricing: {
117
- input: {
118
- normal: 0.2,
119
- cached: 0.05,
120
- },
121
- output: {
122
- normal: 0.5,
123
- },
124
- },
125
- } as const satisfies ModelMeta
126
-
127
- const GROK_4 = {
128
- name: 'grok-4',
129
- context_window: 256_000,
130
- supports: {
131
- input: ['text', 'image'],
132
- output: ['text'],
133
- capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
134
- tools: [] as const,
135
- },
136
- pricing: {
137
- input: {
138
- normal: 3,
139
- cached: 0.75,
140
- },
141
- output: {
142
- normal: 15,
143
- },
144
- },
145
- } as const satisfies ModelMeta
146
-
147
- const GROK_3_MINI = {
148
- name: 'grok-3-mini',
149
- context_window: 131_072,
150
- supports: {
151
- input: ['text'],
152
- output: ['text'],
153
- capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
154
- tools: [] as const,
155
- },
156
- pricing: {
157
- input: {
158
- normal: 0.3,
159
- cached: 0.075,
160
- },
161
- output: {
162
- normal: 0.5,
163
- },
164
- },
165
- } as const satisfies ModelMeta
166
-
167
- const GROK_3 = {
168
- name: 'grok-3',
169
- context_window: 131_072,
170
- supports: {
171
- input: ['text'],
172
- output: ['text'],
173
- capabilities: ['structured_outputs', 'tool_calling'],
174
- tools: [] as const,
175
- },
176
- pricing: {
177
- input: {
178
- normal: 3,
179
- cached: 0.75,
180
- },
181
- output: {
182
- normal: 15,
183
- },
184
- },
185
- } as const satisfies ModelMeta
186
-
187
- const GROK_2_VISION = {
188
- name: 'grok-2-vision-1212',
189
- context_window: 32_768,
190
- supports: {
191
- input: ['text', 'image'],
192
- output: ['text'],
193
- capabilities: ['structured_outputs', 'tool_calling'],
194
- tools: [] as const,
195
- },
196
- pricing: {
197
- input: {
198
- normal: 2,
199
- },
200
- output: {
201
- normal: 10,
202
- },
203
- },
204
- } as const satisfies ModelMeta
38
+ const GROK_RESPONSES_TOOLS = [
39
+ 'web_search',
40
+ 'x_search',
41
+ 'file_search',
42
+ 'mcp',
43
+ ] as const satisfies ReadonlyArray<GrokProviderToolKind>
205
44
 
206
45
  const GROK_2_IMAGE = {
207
46
  name: 'grok-2-image-1212',
@@ -252,46 +91,43 @@ const GROK_IMAGINE_IMAGE_QUALITY = {
252
91
  },
253
92
  } as const satisfies ModelMeta
254
93
 
255
- /**
256
- * Grok Chat Models
257
- * Based on xAI's available models as of 2025
258
- */
259
- const GROK_4_20 = {
260
- name: 'grok-4.20',
261
- context_window: 2_000_000,
94
+ // Imagine API video models. Pricing is per second of generated video
95
+ // (output only); generated videos carry an audio track.
96
+ //
97
+ // grok-imagine-video (v1.0) supports both text-to-video (a starting image is
98
+ // optional) and image-to-video. grok-imagine-video-1.5 is image-to-video
99
+ // only: a starting-frame image is required (the text prompt describes the
100
+ // desired motion) — its text-to-video is rejected by the API.
101
+ const GROK_IMAGINE_VIDEO = {
102
+ name: 'grok-imagine-video',
262
103
  supports: {
263
- input: ['text', 'image', 'document'],
264
- output: ['text'],
265
- capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
266
- tools: [] as const,
104
+ input: ['text', 'image'],
105
+ output: ['video', 'audio'],
267
106
  },
268
107
  pricing: {
269
108
  input: {
270
- normal: 2,
271
- cached: 0.2,
109
+ normal: 0,
272
110
  },
273
111
  output: {
274
- normal: 6,
112
+ // per second of video
113
+ normal: 0.05,
275
114
  },
276
115
  },
277
116
  } as const satisfies ModelMeta
278
117
 
279
- const GROK_4_20_MULTI_AGENT = {
280
- name: 'grok-4.20-multi-agent',
281
- context_window: 2_000_000,
118
+ const GROK_IMAGINE_VIDEO_1_5 = {
119
+ name: 'grok-imagine-video-1.5',
282
120
  supports: {
283
- input: ['text', 'image', 'document'],
284
- output: ['text'],
285
- capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
286
- tools: [] as const,
121
+ input: ['text', 'image'],
122
+ output: ['video', 'audio'],
287
123
  },
288
124
  pricing: {
289
125
  input: {
290
- normal: 2,
291
- cached: 0.2,
126
+ normal: 0,
292
127
  },
293
128
  output: {
294
- normal: 6,
129
+ // per second of video
130
+ normal: 0.08,
295
131
  },
296
132
  },
297
133
  } as const satisfies ModelMeta
@@ -303,7 +139,7 @@ const GROK_4_3 = {
303
139
  input: ['text', 'image'],
304
140
  output: ['text'],
305
141
  capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
306
- tools: [],
142
+ tools: GROK_RESPONSES_TOOLS,
307
143
  },
308
144
  pricing: {
309
145
  input: {
@@ -323,7 +159,7 @@ const GROK_BUILD_0_1 = {
323
159
  input: ['text', 'image'],
324
160
  output: ['text'],
325
161
  capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],
326
- tools: [],
162
+ tools: GROK_RESPONSES_TOOLS,
327
163
  },
328
164
  pricing: {
329
165
  input: {
@@ -336,48 +172,10 @@ const GROK_BUILD_0_1 = {
336
172
  },
337
173
  } as const satisfies ModelMeta
338
174
 
339
- export const GROK_CHAT_MODELS = [
340
- GROK_4_1_FAST_REASONING.name,
341
- GROK_4_1_FAST_NON_REASONING.name,
342
- GROK_CODE_FAST_1.name,
343
- GROK_4_FAST_REASONING.name,
344
- GROK_4_FAST_NON_REASONING.name,
345
- GROK_4.name,
346
- GROK_3.name,
347
- GROK_3_MINI.name,
348
- GROK_2_VISION.name,
349
-
350
- GROK_4_20.name,
351
- GROK_4_20_MULTI_AGENT.name,
352
-
353
- GROK_4_3.name,
354
-
355
- GROK_BUILD_0_1.name,
356
- ] as const
357
-
358
175
  /**
359
- * Grok models that support combining `tools` + `response_format: json_schema`
360
- * in a single streaming Chat Completions request (per issue #605). xAI
361
- * docs gate this to the Grok 4 family — Grok 2 / 3 reject the
362
- * combination. Grok 2 image generation is not a chat model, omitted.
363
- *
364
- * Note: Grok streams tool-call arguments atomically (not token-streamed)
365
- * per the issue's source matrix; partial-JSON tool-arg parsing should be
366
- * skipped for Grok specifically. That's a separate adapter concern from
367
- * this set — the set only gates whether the engine takes the native
368
- * combined path vs the legacy finalization path.
176
+ * Grok chat models supported by the Responses adapter.
369
177
  */
370
- export const GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS = new Set<string>([
371
- GROK_4_1_FAST_REASONING.name,
372
- GROK_4_1_FAST_NON_REASONING.name,
373
- GROK_CODE_FAST_1.name,
374
- GROK_4_FAST_REASONING.name,
375
- GROK_4_FAST_NON_REASONING.name,
376
- GROK_4.name,
377
- GROK_4_20.name,
378
- GROK_4_20_MULTI_AGENT.name,
379
- GROK_4_3.name,
380
- ])
178
+ export const GROK_CHAT_MODELS = [GROK_BUILD_0_1.name, GROK_4_3.name] as const
381
179
 
382
180
  /**
383
181
  * Grok Image Generation Models
@@ -388,6 +186,16 @@ export const GROK_IMAGE_MODELS = [
388
186
  GROK_IMAGINE_IMAGE_QUALITY.name,
389
187
  ] as const
390
188
 
189
+ /**
190
+ * Grok Video Generation Models (xAI Imagine API)
191
+ *
192
+ * @experimental Video generation is an experimental feature and may change.
193
+ */
194
+ export const GROK_VIDEO_MODELS = [
195
+ GROK_IMAGINE_VIDEO.name,
196
+ GROK_IMAGINE_VIDEO_1_5.name,
197
+ ] as const
198
+
391
199
  // xAI's `/v1/tts` endpoint is endpoint-addressed and does not take a `model`
392
200
  // parameter. This synthetic identifier satisfies the SDK's `TTSOptions.model`
393
201
  // contract and provides a stable value for logging and fixture matching.
@@ -441,6 +249,7 @@ export const GROK_REALTIME_MODELS = [
441
249
 
442
250
  export type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]
443
251
  export type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]
252
+ export type GrokVideoModel = (typeof GROK_VIDEO_MODELS)[number]
444
253
  export type GrokTTSModel = (typeof GROK_TTS_MODELS)[number]
445
254
  export type GrokTranscriptionModel = (typeof GROK_TRANSCRIPTION_MODELS)[number]
446
255
  export type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]
@@ -450,68 +259,28 @@ export type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]
450
259
  * Used for type inference when constructing multimodal messages.
451
260
  */
452
261
  export type GrokModelInputModalitiesByName = {
453
- [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.input
454
- [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.input
455
- [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.input
456
- [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.input
457
- [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.input
458
- [GROK_4.name]: typeof GROK_4.supports.input
459
- [GROK_3.name]: typeof GROK_3.supports.input
460
- [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.input
461
- [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.input
462
- [GROK_4_20.name]: typeof GROK_4_20.supports.input
463
- [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.input
464
262
  [GROK_4_3.name]: typeof GROK_4_3.supports.input
465
263
  [GROK_BUILD_0_1.name]: typeof GROK_BUILD_0_1.supports.input
466
264
  }
467
265
 
468
- /**
469
- * Type-only map from Grok chat model name to its provider options type.
470
- * Since Grok uses OpenAI-compatible API, we reuse OpenAI provider options.
471
- */
472
- export type GrokChatModelProviderOptionsByName = {
473
- [K in (typeof GROK_CHAT_MODELS)[number]]: GrokProviderOptions
474
- }
475
-
476
266
  /**
477
267
  * Type-only map from Grok chat model name to its supported provider tools.
478
- * Grok exposes no provider-specific tool factories, so every model gets an
479
- * empty tuple. This ensures that passing an Anthropic/OpenAI ProviderTool to
480
- * a Grok adapter produces a compile-time type error.
268
+ * Keeps Grok provider-tool factories type-checked against the models that
269
+ * advertise xAI Responses server-side tools.
481
270
  */
482
271
  export type GrokChatModelToolCapabilitiesByName = {
483
- [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.tools
484
- [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.tools
485
- [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.tools
486
- [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.tools
487
- [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.tools
488
- [GROK_4.name]: typeof GROK_4.supports.tools
489
- [GROK_3.name]: typeof GROK_3.supports.tools
490
- [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.tools
491
- [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.tools
492
- [GROK_4_20.name]: typeof GROK_4_20.supports.tools
493
- [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.tools
272
+ [GROK_4_3.name]: typeof GROK_4_3.supports.tools
273
+ [GROK_BUILD_0_1.name]: typeof GROK_BUILD_0_1.supports.tools
494
274
  }
495
275
 
276
+ export type GrokProviderOptions = GrokTextProviderOptions
277
+
496
278
  /**
497
- * Grok-specific provider options
498
- * Based on OpenAI-compatible API options
279
+ * Type-only map from Grok chat model name to its provider options type.
499
280
  */
500
- export interface GrokProviderOptions {
501
- /** Temperature for response generation (0-2) */
502
- temperature?: number
503
- /** Maximum tokens in the response */
504
- max_tokens?: number
505
- /** Top-p sampling parameter */
506
- top_p?: number
507
- /** Frequency penalty (-2.0 to 2.0) */
508
- frequency_penalty?: number
509
- /** Presence penalty (-2.0 to 2.0) */
510
- presence_penalty?: number
511
- /** Stop sequences */
512
- stop?: string | Array<string>
513
- /** A unique identifier representing your end-user */
514
- user?: string
281
+ export type GrokChatModelProviderOptionsByName = {
282
+ [GROK_4_3.name]: GrokProviderOptions
283
+ [GROK_BUILD_0_1.name]: GrokBuildProviderOptions
515
284
  }
516
285
 
517
286
  // ===========================
@@ -1,10 +1,22 @@
1
1
  /**
2
2
  * Grok Text Provider Options
3
3
  *
4
- * Grok uses an OpenAI-compatible Chat Completions API.
5
- * However, not all OpenAI features may be supported by Grok.
4
+ * Grok uses xAI's OpenAI-compatible Responses API. Engine-managed fields
5
+ * such as `model`, `input`, `tools`, and `text.format` are owned by the
6
+ * adapter; user-supplied values live under `modelOptions`.
6
7
  */
7
8
 
9
+ import type { ResponseCreateParams } from 'openai/resources/responses/responses'
10
+
11
+ export type GrokReasoningEffort = 'none' | 'low' | 'medium' | 'high'
12
+
13
+ export type GrokReasoning = Omit<
14
+ NonNullable<ResponseCreateParams['reasoning']>,
15
+ 'effort'
16
+ > & {
17
+ effort?: GrokReasoningEffort
18
+ }
19
+
8
20
  /**
9
21
  * Base provider options for Grok text/chat models
10
22
  */
@@ -18,9 +30,10 @@ export interface GrokBaseOptions {
18
30
 
19
31
  /**
20
32
  * Grok-specific provider options for text/chat
21
- * Based on OpenAI-compatible API options
33
+ * Based on xAI Responses API options
22
34
  */
23
- export interface GrokTextProviderOptions extends GrokBaseOptions {
35
+ export interface GrokTextProviderOptions
36
+ extends GrokBaseOptions, Record<string, unknown> {
24
37
  /**
25
38
  * Temperature for response generation (0-2)
26
39
  * Higher values make output more random, lower values more focused
@@ -34,19 +47,26 @@ export interface GrokTextProviderOptions extends GrokBaseOptions {
34
47
  /**
35
48
  * Maximum tokens in the response
36
49
  */
37
- max_tokens?: number
50
+ max_output_tokens?: number
38
51
  /**
39
- * Frequency penalty (-2.0 to 2.0)
52
+ * Whether xAI should store the response. Defaults to `false` in the adapter.
40
53
  */
41
- frequency_penalty?: number
54
+ store?: boolean
42
55
  /**
43
- * Presence penalty (-2.0 to 2.0)
56
+ * Additional response fields to include. Defaults to encrypted reasoning.
44
57
  */
45
- presence_penalty?: number
58
+ include?: ResponseCreateParams['include']
46
59
  /**
47
- * Stop sequences
60
+ * xAI/OpenAI-compatible reasoning controls for reasoning-capable models.
48
61
  */
49
- stop?: string | Array<string>
62
+ reasoning?: GrokReasoning
63
+ }
64
+
65
+ export type GrokBuildProviderOptions = Omit<
66
+ GrokTextProviderOptions,
67
+ 'reasoning'
68
+ > & {
69
+ reasoning?: never
50
70
  }
51
71
 
52
72
  /**