@tanstack/ai-grok 0.6.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/dist/esm/adapters/image.js +36 -17
  2. package/dist/esm/adapters/image.js.map +1 -1
  3. package/dist/esm/adapters/summarize.js +51 -22
  4. package/dist/esm/adapters/summarize.js.map +1 -1
  5. package/dist/esm/adapters/text.js +25 -10
  6. package/dist/esm/adapters/text.js.map +1 -1
  7. package/dist/esm/adapters/transcription.d.ts +84 -0
  8. package/dist/esm/adapters/transcription.js +109 -0
  9. package/dist/esm/adapters/transcription.js.map +1 -0
  10. package/dist/esm/adapters/tts.d.ts +70 -0
  11. package/dist/esm/adapters/tts.js +137 -0
  12. package/dist/esm/adapters/tts.js.map +1 -0
  13. package/dist/esm/audio/transcription-provider-options.d.ts +41 -0
  14. package/dist/esm/audio/tts-provider-options.d.ts +42 -0
  15. package/dist/esm/index.d.ts +8 -2
  16. package/dist/esm/index.js +17 -2
  17. package/dist/esm/index.js.map +1 -1
  18. package/dist/esm/model-meta.d.ts +6 -0
  19. package/dist/esm/model-meta.js +22 -1
  20. package/dist/esm/model-meta.js.map +1 -1
  21. package/dist/esm/realtime/adapter.d.ts +21 -0
  22. package/dist/esm/realtime/adapter.js +816 -0
  23. package/dist/esm/realtime/adapter.js.map +1 -0
  24. package/dist/esm/realtime/index.d.ts +4 -0
  25. package/dist/esm/realtime/realtime-contract.d.ts +30 -0
  26. package/dist/esm/realtime/token.d.ts +22 -0
  27. package/dist/esm/realtime/token.js +73 -0
  28. package/dist/esm/realtime/token.js.map +1 -0
  29. package/dist/esm/realtime/types.d.ts +95 -0
  30. package/dist/esm/utils/audio.d.ts +23 -0
  31. package/dist/esm/utils/audio.js +171 -0
  32. package/dist/esm/utils/audio.js.map +1 -0
  33. package/dist/esm/utils/index.d.ts +1 -0
  34. package/package.json +6 -3
  35. package/src/adapters/image.ts +41 -19
  36. package/src/adapters/summarize.ts +56 -25
  37. package/src/adapters/text.ts +26 -9
  38. package/src/adapters/transcription.ts +233 -0
  39. package/src/adapters/tts.ts +260 -0
  40. package/src/audio/transcription-provider-options.ts +54 -0
  41. package/src/audio/tts-provider-options.ts +44 -0
  42. package/src/index.ts +50 -1
  43. package/src/model-meta.ts +54 -0
  44. package/src/realtime/adapter.ts +1215 -0
  45. package/src/realtime/index.ts +18 -0
  46. package/src/realtime/realtime-contract.ts +46 -0
  47. package/src/realtime/token.ts +131 -0
  48. package/src/realtime/types.ts +105 -0
  49. package/src/utils/audio.ts +217 -0
  50. package/src/utils/index.ts +1 -0
@@ -0,0 +1,260 @@
1
+ import { BaseTTSAdapter } from '@tanstack/ai/adapters'
2
+ import { arrayBufferToBase64, generateId, getGrokApiKeyFromEnv } from '../utils'
3
+ import type { TTSOptions, TTSResult } from '@tanstack/ai'
4
+ import type { GrokTTSModel } from '../model-meta'
5
+ import type {
6
+ GrokTTSCodec,
7
+ GrokTTSProviderOptions,
8
+ GrokTTSVoice,
9
+ } from '../audio/tts-provider-options'
10
+
11
+ const DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'
12
+
13
+ /**
14
+ * Configuration for the Grok TTS adapter.
15
+ *
16
+ * Unlike chat/image/summarize adapters, TTS does not use the OpenAI SDK
17
+ * because xAI's `/v1/tts` endpoint is not OpenAI-compatible. This config
18
+ * is a minimal subset suitable for direct `fetch` calls.
19
+ */
20
+ export interface GrokSpeechConfig {
21
+ apiKey: string
22
+ baseURL?: string
23
+ /** Additional headers to merge into every request (e.g., test IDs). */
24
+ defaultHeaders?: Record<string, string>
25
+ }
26
+
27
+ /**
28
+ * Grok Text-to-Speech Adapter.
29
+ *
30
+ * Talks to `POST {baseURL}/tts` per
31
+ * https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
32
+ */
33
+ export class GrokSpeechAdapter<
34
+ TModel extends GrokTTSModel,
35
+ > extends BaseTTSAdapter<TModel, GrokTTSProviderOptions> {
36
+ readonly name = 'grok' as const
37
+
38
+ private readonly apiKey: string
39
+ private readonly baseURL: string
40
+ private readonly defaultHeaders: Record<string, string>
41
+
42
+ constructor(config: GrokSpeechConfig, model: TModel) {
43
+ super(model, config)
44
+ this.apiKey = config.apiKey
45
+ this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, '')
46
+ this.defaultHeaders = config.defaultHeaders ?? {}
47
+ }
48
+
49
+ async generateSpeech(
50
+ options: TTSOptions<GrokTTSProviderOptions>,
51
+ ): Promise<TTSResult> {
52
+ const { logger } = options
53
+ const { model, text, voice, format, modelOptions } = options
54
+
55
+ logger.request(`activity=generateSpeech provider=grok model=${model}`, {
56
+ provider: 'grok',
57
+ model,
58
+ })
59
+
60
+ const { body, codec, sampleRateForContentType } = buildTTSRequestBody({
61
+ text,
62
+ voice,
63
+ format,
64
+ modelOptions,
65
+ })
66
+
67
+ try {
68
+ const response = await fetch(`${this.baseURL}/tts`, {
69
+ method: 'POST',
70
+ headers: {
71
+ // `defaultHeaders` first so the adapter's Authorization / Content-Type
72
+ // always win — otherwise a caller-supplied `Authorization` header
73
+ // could silently clobber the bearer token.
74
+ ...this.defaultHeaders,
75
+ Authorization: `Bearer ${this.apiKey}`,
76
+ 'Content-Type': 'application/json',
77
+ },
78
+ body: JSON.stringify(body),
79
+ })
80
+
81
+ if (!response.ok) {
82
+ const errorText = await response.text()
83
+ throw new Error(
84
+ `Grok TTS request failed: ${response.status} ${errorText}`,
85
+ )
86
+ }
87
+
88
+ const arrayBuffer = await response.arrayBuffer()
89
+ const audio = arrayBufferToBase64(arrayBuffer)
90
+
91
+ return {
92
+ id: generateId(this.name),
93
+ model,
94
+ audio,
95
+ format: codec,
96
+ contentType: getContentType(codec, sampleRateForContentType),
97
+ }
98
+ } catch (error) {
99
+ logger.errors('grok.generateSpeech fatal', {
100
+ error,
101
+ source: 'grok.generateSpeech',
102
+ })
103
+ throw error
104
+ }
105
+ }
106
+ }
107
+
108
+ /**
109
+ * Build the JSON body for `POST /v1/tts`, resolving codec / sample-rate / voice
110
+ * defaults in one place.
111
+ *
112
+ * Returns the request `body`, the resolved `codec`, and the `sampleRateForContentType`
113
+ * used by the caller to label the response via `getContentType`.
114
+ */
115
+ export function buildTTSRequestBody(options: {
116
+ text: string
117
+ voice: string | undefined
118
+ format: TTSOptions['format'] | undefined
119
+ modelOptions: GrokTTSProviderOptions | undefined
120
+ }): {
121
+ body: Record<string, unknown>
122
+ codec: GrokTTSCodec
123
+ sampleRateForContentType: number
124
+ } {
125
+ const { text, voice, format, modelOptions } = options
126
+
127
+ const codec = pickCodec(modelOptions?.codec, format)
128
+
129
+ // Only forward `sample_rate` when either:
130
+ // - the caller explicitly set `modelOptions.sample_rate`, or
131
+ // - the codec's Content-Type carries the rate (pcm → audio/L16;rate=…).
132
+ // For mp3/wav/opus/aac/flac we leave sample_rate unset so xAI's server
133
+ // default applies.
134
+ const callerSampleRate = modelOptions?.sample_rate
135
+ // Default sample rate documented in GrokTTSProviderOptions is 24000 Hz —
136
+ // used only when we MUST attach a rate to the contentType (pcm) and the
137
+ // caller didn't pick one.
138
+ const pcmDefault = 24000
139
+ const needsRateInContentType = codec === 'pcm'
140
+
141
+ const outputFormat: Record<string, unknown> = { codec }
142
+ if (callerSampleRate !== undefined) {
143
+ outputFormat.sample_rate = callerSampleRate
144
+ } else if (needsRateInContentType) {
145
+ outputFormat.sample_rate = pcmDefault
146
+ }
147
+ if (codec === 'mp3' && modelOptions?.bit_rate !== undefined) {
148
+ outputFormat.bit_rate = modelOptions.bit_rate
149
+ }
150
+
151
+ // pcm embeds the rate in `audio/L16;rate=…`; mulaw/alaw embed it in
152
+ // `audio/PCMU;rate=…` / `audio/PCMA;rate=…` when non-default. mp3/wav
153
+ // don't carry a rate parameter so the value is unused for those.
154
+ const sampleRateForContentType = callerSampleRate ?? pcmDefault
155
+
156
+ const body: Record<string, unknown> = {
157
+ text,
158
+ voice_id: (voice as GrokTTSVoice | undefined) ?? 'eve',
159
+ language: modelOptions?.language ?? 'en',
160
+ output_format: outputFormat,
161
+ }
162
+ if (modelOptions?.optimize_streaming_latency !== undefined) {
163
+ body.optimize_streaming_latency = modelOptions.optimize_streaming_latency
164
+ }
165
+ if (modelOptions?.text_normalization !== undefined) {
166
+ body.text_normalization = modelOptions.text_normalization
167
+ }
168
+
169
+ return { body, codec, sampleRateForContentType }
170
+ }
171
+
172
+ /**
173
+ * Maps the cross-provider `TTSOptions.format` onto Grok's supported codecs.
174
+ * `opus`, `aac`, and `flac` are not supported by xAI TTS (which only exposes
175
+ * mp3/wav/pcm/mulaw/alaw) — we fall back to mp3. An explicit
176
+ * `modelOptions.codec` always wins.
177
+ */
178
+ function pickCodec(
179
+ codecOverride: GrokTTSCodec | undefined,
180
+ format: TTSOptions['format'] | undefined,
181
+ ): GrokTTSCodec {
182
+ if (codecOverride) return codecOverride
183
+ if (!format) return 'mp3'
184
+ switch (format) {
185
+ case 'mp3':
186
+ case 'wav':
187
+ case 'pcm':
188
+ return format
189
+ case 'flac':
190
+ case 'opus':
191
+ case 'aac':
192
+ return 'mp3'
193
+ default:
194
+ return 'mp3'
195
+ }
196
+ }
197
+
198
+ export function getContentType(
199
+ codec: GrokTTSCodec,
200
+ sampleRate: number,
201
+ ): string {
202
+ switch (codec) {
203
+ case 'mp3':
204
+ return 'audio/mpeg'
205
+ case 'wav':
206
+ return 'audio/wav'
207
+ case 'pcm':
208
+ // `audio/L16` requires a `rate` parameter per RFC 3551/3555.
209
+ return `audio/L16;rate=${sampleRate}`
210
+ case 'mulaw':
211
+ // `audio/basic` is 8 kHz mono by RFC 2046 registration. For non-8kHz
212
+ // streams xAI still produces mulaw-encoded bytes at the requested
213
+ // rate, but the registered MIME can't carry that rate — so we use
214
+ // the non-standard but commonly-supported `audio/PCMU;rate=…` (RFC 3551
215
+ // RTP payload name) whenever the caller asked for a rate other than
216
+ // 8000, and keep `audio/basic` for the standard 8kHz case.
217
+ return sampleRate === 8000
218
+ ? 'audio/basic'
219
+ : `audio/PCMU;rate=${sampleRate}`
220
+ case 'alaw':
221
+ return sampleRate === 8000
222
+ ? 'audio/x-alaw-basic'
223
+ : `audio/PCMA;rate=${sampleRate}`
224
+ }
225
+ }
226
+
227
+ /**
228
+ * Creates a Grok speech (TTS) adapter with an explicit API key.
229
+ *
230
+ * @example
231
+ * ```typescript
232
+ * const adapter = createGrokSpeech('grok-tts', 'xai-...')
233
+ * const result = await generateSpeech({
234
+ * adapter,
235
+ * text: 'Hello from Grok',
236
+ * voice: 'eve',
237
+ * })
238
+ * ```
239
+ */
240
+ export function createGrokSpeech<TModel extends GrokTTSModel>(
241
+ model: TModel,
242
+ apiKey: string,
243
+ config?: Omit<GrokSpeechConfig, 'apiKey'>,
244
+ ): GrokSpeechAdapter<TModel> {
245
+ return new GrokSpeechAdapter({ apiKey, ...config }, model)
246
+ }
247
+
248
+ /**
249
+ * Creates a Grok speech (TTS) adapter, reading the API key from
250
+ * `XAI_API_KEY` in the environment.
251
+ *
252
+ * @throws Error if `XAI_API_KEY` is not set.
253
+ */
254
+ export function grokSpeech<TModel extends GrokTTSModel>(
255
+ model: TModel,
256
+ config?: Omit<GrokSpeechConfig, 'apiKey'>,
257
+ ): GrokSpeechAdapter<TModel> {
258
+ const apiKey = getGrokApiKeyFromEnv()
259
+ return createGrokSpeech(model, apiKey, config)
260
+ }
@@ -0,0 +1,54 @@
1
+ /**
2
+ * Grok STT supported audio formats.
3
+ * See https://docs.x.ai/developers/rest-api-reference/inference/voice
4
+ */
5
+ export type GrokSTTAudioFormat =
6
+ | 'pcm'
7
+ | 'mulaw'
8
+ | 'alaw'
9
+ | 'wav'
10
+ | 'mp3'
11
+ | 'ogg'
12
+ | 'opus'
13
+ | 'flac'
14
+ | 'aac'
15
+ | 'mp4'
16
+ | 'm4a'
17
+ | 'mkv'
18
+
19
+ /**
20
+ * Provider-specific options for Grok transcription (`POST /v1/stt`).
21
+ */
22
+ export interface GrokTranscriptionProviderOptions {
23
+ /**
24
+ * The format of the provided audio. Required for raw codecs (pcm, mulaw, alaw).
25
+ */
26
+ audio_format?: GrokSTTAudioFormat
27
+ /**
28
+ * Sample rate of the audio (Hz). Required for raw codecs.
29
+ */
30
+ sample_rate?: number
31
+ /**
32
+ * Apply inverse text normalization (e.g. "one hundred" → "100"). Requires
33
+ * `language` to be set on the core `TranscriptionOptions`.
34
+ *
35
+ * NOTE: xAI's STT API exposes this on the wire as `format` (a boolean
36
+ * toggle). We surface it under the clearer name
37
+ * `inverse_text_normalization` on the SDK, and translate to the wire name
38
+ * inside the adapter.
39
+ */
40
+ inverse_text_normalization?: boolean
41
+ /**
42
+ * Treat the audio as multichannel. When enabled, `channels` must also be set.
43
+ */
44
+ multichannel?: boolean
45
+ /**
46
+ * Channel count for multichannel raw audio (2–8).
47
+ */
48
+ channels?: number
49
+ /**
50
+ * Enable speaker diarization. When true, response words include a `speaker`
51
+ * field.
52
+ */
53
+ diarize?: boolean
54
+ }
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Grok TTS voice options.
3
+ * See https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
4
+ */
5
+ export type GrokTTSVoice = 'eve' | 'ara' | 'rex' | 'sal' | 'leo'
6
+
7
+ /**
8
+ * Grok TTS output audio codecs.
9
+ * Grok does NOT support opus or aac; those formats are mapped to mp3.
10
+ */
11
+ export type GrokTTSCodec = 'mp3' | 'wav' | 'pcm' | 'mulaw' | 'alaw'
12
+
13
+ /**
14
+ * Provider-specific options for Grok TTS (`POST /v1/tts`).
15
+ */
16
+ export interface GrokTTSProviderOptions {
17
+ /**
18
+ * BCP-47 language code (e.g., `en`, `zh`, `pt-BR`) or `'auto'` for detection.
19
+ * Defaults to `'en'` when not provided.
20
+ */
21
+ language?: string
22
+ /**
23
+ * Audio codec. Overrides the `format` field on `TTSOptions` when set.
24
+ */
25
+ codec?: GrokTTSCodec
26
+ /**
27
+ * Sample rate in Hz. Valid values: 8000, 16000, 22050, 24000, 44100, 48000.
28
+ * Defaults to 24000.
29
+ */
30
+ sample_rate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000
31
+ /**
32
+ * Bit rate for MP3 output. Ignored for other codecs.
33
+ * Valid values: 32000, 64000, 96000, 128000, 192000. Defaults to 128000.
34
+ */
35
+ bit_rate?: 32000 | 64000 | 96000 | 128000 | 192000
36
+ /**
37
+ * Set to 1 for lower latency streaming; 0 (default) for normal quality.
38
+ */
39
+ optimize_streaming_latency?: 0 | 1
40
+ /**
41
+ * Enable text normalization. Defaults to false.
42
+ */
43
+ text_normalization?: boolean
44
+ }
package/src/index.ts CHANGED
@@ -33,6 +33,31 @@ export type {
33
33
  GrokImageModelProviderOptionsByName,
34
34
  } from './image/image-provider-options'
35
35
 
36
+ // Speech (TTS) adapter - for text-to-speech
37
+ export {
38
+ GrokSpeechAdapter,
39
+ createGrokSpeech,
40
+ grokSpeech,
41
+ type GrokSpeechConfig,
42
+ } from './adapters/tts'
43
+ export type {
44
+ GrokTTSProviderOptions,
45
+ GrokTTSVoice,
46
+ GrokTTSCodec,
47
+ } from './audio/tts-provider-options'
48
+
49
+ // Transcription adapter - for speech-to-text
50
+ export {
51
+ GrokTranscriptionAdapter,
52
+ createGrokTranscription,
53
+ grokTranscription,
54
+ type GrokTranscriptionConfig,
55
+ } from './adapters/transcription'
56
+ export type {
57
+ GrokTranscriptionProviderOptions,
58
+ GrokSTTAudioFormat,
59
+ } from './audio/transcription-provider-options'
60
+
36
61
  // ============================================================================
37
62
  // Type Exports
38
63
  // ============================================================================
@@ -45,8 +70,17 @@ export type {
45
70
  ResolveInputModalities,
46
71
  GrokChatModel,
47
72
  GrokImageModel,
73
+ GrokTTSModel,
74
+ GrokTranscriptionModel,
75
+ GrokRealtimeModel,
76
+ } from './model-meta'
77
+ export {
78
+ GROK_CHAT_MODELS,
79
+ GROK_IMAGE_MODELS,
80
+ GROK_TTS_MODELS,
81
+ GROK_TRANSCRIPTION_MODELS,
82
+ GROK_REALTIME_MODELS,
48
83
  } from './model-meta'
49
- export { GROK_CHAT_MODELS, GROK_IMAGE_MODELS } from './model-meta'
50
84
  export type {
51
85
  GrokTextMetadata,
52
86
  GrokImageMetadata,
@@ -55,3 +89,18 @@ export type {
55
89
  GrokDocumentMetadata,
56
90
  GrokMessageMetadataByModality,
57
91
  } from './message-types'
92
+
93
+ // ============================================================================
94
+ // Realtime (Voice Agent) Adapters
95
+ // ============================================================================
96
+
97
+ export { grokRealtimeToken, grokRealtime } from './realtime/index'
98
+
99
+ export type {
100
+ GrokRealtimeVoice,
101
+ GrokRealtimeTokenOptions,
102
+ GrokRealtimeOptions,
103
+ GrokTurnDetection,
104
+ GrokSemanticVADConfig,
105
+ GrokServerVADConfig,
106
+ } from './realtime/index'
package/src/model-meta.ts CHANGED
@@ -283,8 +283,62 @@ export const GROK_CHAT_MODELS = [
283
283
  */
284
284
  export const GROK_IMAGE_MODELS = [GROK_2_IMAGE.name] as const
285
285
 
286
+ // xAI's `/v1/tts` endpoint is endpoint-addressed and does not take a `model`
287
+ // parameter. This synthetic identifier satisfies the SDK's `TTSOptions.model`
288
+ // contract and provides a stable value for logging and fixture matching.
289
+ const GROK_TTS = {
290
+ name: 'grok-tts',
291
+ supports: {
292
+ input: ['text'],
293
+ output: ['audio'],
294
+ },
295
+ } as const satisfies ModelMeta
296
+
297
+ // xAI's `/v1/stt` endpoint is endpoint-addressed and does not take a `model`
298
+ // parameter. This synthetic identifier satisfies the SDK's
299
+ // `TranscriptionOptions.model` contract.
300
+ const GROK_STT = {
301
+ name: 'grok-stt',
302
+ supports: {
303
+ input: ['audio'],
304
+ output: ['text'],
305
+ },
306
+ } as const satisfies ModelMeta
307
+
308
+ const GROK_VOICE_FAST_1 = {
309
+ name: 'grok-voice-fast-1.0',
310
+ supports: {
311
+ input: ['audio', 'text'],
312
+ output: ['audio', 'text'],
313
+ capabilities: ['tool_calling'],
314
+ tools: [] as const,
315
+ },
316
+ } as const satisfies ModelMeta
317
+
318
+ const GROK_VOICE_THINK_FAST_1 = {
319
+ name: 'grok-voice-think-fast-1.0',
320
+ supports: {
321
+ input: ['audio', 'text'],
322
+ output: ['audio', 'text'],
323
+ capabilities: ['reasoning', 'tool_calling'],
324
+ tools: [] as const,
325
+ },
326
+ } as const satisfies ModelMeta
327
+
328
+ export const GROK_TTS_MODELS = [GROK_TTS.name] as const
329
+
330
+ export const GROK_TRANSCRIPTION_MODELS = [GROK_STT.name] as const
331
+
332
+ export const GROK_REALTIME_MODELS = [
333
+ GROK_VOICE_FAST_1.name,
334
+ GROK_VOICE_THINK_FAST_1.name,
335
+ ] as const
336
+
286
337
  export type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]
287
338
  export type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]
339
+ export type GrokTTSModel = (typeof GROK_TTS_MODELS)[number]
340
+ export type GrokTranscriptionModel = (typeof GROK_TRANSCRIPTION_MODELS)[number]
341
+ export type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]
288
342
 
289
343
  /**
290
344
  * Type-only map from Grok chat model name to its supported input modalities.