@tanstack/ai-grok 0.6.8 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/dist/esm/adapters/image.js +15 -8
  2. package/dist/esm/adapters/image.js.map +1 -1
  3. package/dist/esm/adapters/transcription.d.ts +84 -0
  4. package/dist/esm/adapters/transcription.js +109 -0
  5. package/dist/esm/adapters/transcription.js.map +1 -0
  6. package/dist/esm/adapters/tts.d.ts +70 -0
  7. package/dist/esm/adapters/tts.js +137 -0
  8. package/dist/esm/adapters/tts.js.map +1 -0
  9. package/dist/esm/audio/transcription-provider-options.d.ts +41 -0
  10. package/dist/esm/audio/tts-provider-options.d.ts +42 -0
  11. package/dist/esm/index.d.ts +8 -2
  12. package/dist/esm/index.js +17 -2
  13. package/dist/esm/index.js.map +1 -1
  14. package/dist/esm/model-meta.d.ts +6 -0
  15. package/dist/esm/model-meta.js +22 -1
  16. package/dist/esm/model-meta.js.map +1 -1
  17. package/dist/esm/realtime/adapter.d.ts +21 -0
  18. package/dist/esm/realtime/adapter.js +816 -0
  19. package/dist/esm/realtime/adapter.js.map +1 -0
  20. package/dist/esm/realtime/index.d.ts +4 -0
  21. package/dist/esm/realtime/realtime-contract.d.ts +30 -0
  22. package/dist/esm/realtime/token.d.ts +22 -0
  23. package/dist/esm/realtime/token.js +73 -0
  24. package/dist/esm/realtime/token.js.map +1 -0
  25. package/dist/esm/realtime/types.d.ts +95 -0
  26. package/dist/esm/utils/audio.d.ts +23 -0
  27. package/dist/esm/utils/audio.js +171 -0
  28. package/dist/esm/utils/audio.js.map +1 -0
  29. package/dist/esm/utils/index.d.ts +1 -0
  30. package/package.json +6 -3
  31. package/src/adapters/image.ts +16 -7
  32. package/src/adapters/transcription.ts +233 -0
  33. package/src/adapters/tts.ts +260 -0
  34. package/src/audio/transcription-provider-options.ts +54 -0
  35. package/src/audio/tts-provider-options.ts +44 -0
  36. package/src/index.ts +50 -1
  37. package/src/model-meta.ts +54 -0
  38. package/src/realtime/adapter.ts +1215 -0
  39. package/src/realtime/index.ts +18 -0
  40. package/src/realtime/realtime-contract.ts +46 -0
  41. package/src/realtime/token.ts +131 -0
  42. package/src/realtime/types.ts +105 -0
  43. package/src/utils/audio.ts +217 -0
  44. package/src/utils/index.ts +1 -0
@@ -49,7 +49,7 @@ export class GrokImageAdapter<
49
49
  private client: OpenAI_SDK
50
50
 
51
51
  constructor(config: GrokImageConfig, model: TModel) {
52
- super({}, model)
52
+ super(model, {})
53
53
  this.client = createGrokClient(config)
54
54
  }
55
55
 
@@ -92,12 +92,14 @@ export class GrokImageAdapter<
92
92
  ): OpenAI_SDK.Images.ImageGenerateParams {
93
93
  const { model, prompt, numberOfImages, size, modelOptions } = options
94
94
 
95
+ // Spread modelOptions FIRST so explicit args (model, prompt, n, size) win
96
+ // and user-supplied modelOptions cannot silently override them.
95
97
  return {
98
+ ...modelOptions,
96
99
  model,
97
100
  prompt,
98
101
  n: numberOfImages ?? 1,
99
102
  size: size as OpenAI_SDK.Images.ImageGenerateParams['size'],
100
- ...modelOptions,
101
103
  }
102
104
  }
103
105
 
@@ -105,11 +107,18 @@ export class GrokImageAdapter<
105
107
  model: string,
106
108
  response: OpenAI_SDK.Images.ImagesResponse,
107
109
  ): ImageGenerationResult {
108
- const images: Array<GeneratedImage> = (response.data ?? []).map((item) => ({
109
- b64Json: item.b64_json,
110
- url: item.url,
111
- revisedPrompt: item.revised_prompt,
112
- }))
110
+ const images: Array<GeneratedImage> = (response.data ?? []).flatMap(
111
+ (item): Array<GeneratedImage> => {
112
+ const revisedPrompt = item.revised_prompt
113
+ if (item.b64_json) {
114
+ return [{ b64Json: item.b64_json, revisedPrompt }]
115
+ }
116
+ if (item.url) {
117
+ return [{ url: item.url, revisedPrompt }]
118
+ }
119
+ return []
120
+ },
121
+ )
113
122
 
114
123
  return {
115
124
  id: generateId(this.name),
@@ -0,0 +1,233 @@
1
+ import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
2
+ import { generateId, getGrokApiKeyFromEnv, toAudioFile } from '../utils'
3
+ import type {
4
+ TranscriptionOptions,
5
+ TranscriptionResult,
6
+ TranscriptionWord,
7
+ } from '@tanstack/ai'
8
+ import type { GrokTranscriptionModel } from '../model-meta'
9
+ import type { GrokTranscriptionProviderOptions } from '../audio/transcription-provider-options'
10
+
11
+ /**
12
+ * Grok-specific extension of `TranscriptionWord` that surfaces the extra
13
+ * fields xAI returns when diarization / confidence are enabled. The base
14
+ * cross-provider `TranscriptionWord` contract doesn't include these, so
15
+ * callers who know they're using Grok can narrow with:
16
+ *
17
+ * ```ts
18
+ * const words = result.words as Array<GrokTranscriptionWord> | undefined
19
+ * ```
20
+ */
21
+ export interface GrokTranscriptionWord extends TranscriptionWord {
22
+ /** Model confidence for the word, when xAI returns one. */
23
+ confidence?: number
24
+ /** Speaker index, populated when `modelOptions.diarize === true`. */
25
+ speaker?: number
26
+ }
27
+
28
+ const DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'
29
+
30
+ /**
31
+ * Configuration for the Grok transcription adapter.
32
+ *
33
+ * Uses direct `fetch` rather than the OpenAI SDK because xAI's `/v1/stt`
34
+ * endpoint is not OpenAI-compatible.
35
+ */
36
+ export interface GrokTranscriptionConfig {
37
+ apiKey: string
38
+ baseURL?: string
39
+ /** Additional headers to merge into every request (e.g., test IDs). */
40
+ defaultHeaders?: Record<string, string>
41
+ }
42
+
43
+ /**
44
+ * xAI STT response shape from `POST /v1/stt`.
45
+ * Grok returns word-level timestamps only; no segment array.
46
+ */
47
+ interface GrokSTTWord {
48
+ text: string
49
+ start: number
50
+ end: number
51
+ confidence?: number
52
+ speaker?: number
53
+ }
54
+
55
+ interface GrokSTTResponse {
56
+ text: string
57
+ language?: string
58
+ duration?: number
59
+ words?: Array<GrokSTTWord>
60
+ channels?: Array<unknown>
61
+ }
62
+
63
+ /**
64
+ * Grok Speech-to-Text Adapter.
65
+ *
66
+ * Talks to `POST {baseURL}/stt` per
67
+ * https://docs.x.ai/developers/rest-api-reference/inference/voice
68
+ */
69
+ export class GrokTranscriptionAdapter<
70
+ TModel extends GrokTranscriptionModel,
71
+ > extends BaseTranscriptionAdapter<TModel, GrokTranscriptionProviderOptions> {
72
+ readonly name = 'grok' as const
73
+
74
+ private readonly apiKey: string
75
+ private readonly baseURL: string
76
+ private readonly defaultHeaders: Record<string, string>
77
+
78
+ constructor(config: GrokTranscriptionConfig, model: TModel) {
79
+ super(model, config)
80
+ this.apiKey = config.apiKey
81
+ this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, '')
82
+ this.defaultHeaders = config.defaultHeaders ?? {}
83
+ }
84
+
85
+ async transcribe(
86
+ options: TranscriptionOptions<GrokTranscriptionProviderOptions>,
87
+ ): Promise<TranscriptionResult> {
88
+ const { logger } = options
89
+ const { model, audio, language, modelOptions } = options
90
+
91
+ logger.request(
92
+ `activity=generateTranscription provider=grok model=${model}`,
93
+ { provider: 'grok', model },
94
+ )
95
+
96
+ const file = toAudioFile(audio, modelOptions?.audio_format)
97
+ const form = buildTranscriptionFormData({ file, language, modelOptions })
98
+
99
+ try {
100
+ const response = await fetch(`${this.baseURL}/stt`, {
101
+ method: 'POST',
102
+ headers: {
103
+ // `defaultHeaders` first so Authorization always wins.
104
+ ...this.defaultHeaders,
105
+ Authorization: `Bearer ${this.apiKey}`,
106
+ },
107
+ body: form,
108
+ })
109
+
110
+ if (!response.ok) {
111
+ const errorText = await response.text()
112
+ throw new Error(
113
+ `Grok transcription request failed: ${response.status} ${errorText}`,
114
+ )
115
+ }
116
+
117
+ const data = (await response.json()) as GrokSTTResponse
118
+
119
+ const words: Array<TranscriptionWord> | undefined = data.words?.map(
120
+ (w) => {
121
+ // Construct a GrokTranscriptionWord so that `confidence` and
122
+ // `speaker` (when xAI returns them under `diarize` / confidence
123
+ // mode) are preserved on the result. The returned array is typed
124
+ // as `Array<TranscriptionWord>` per the cross-provider contract;
125
+ // callers who want the extras narrow via `as Array<GrokTranscriptionWord>`.
126
+ const tw: GrokTranscriptionWord = {
127
+ word: w.text,
128
+ start: w.start,
129
+ end: w.end,
130
+ }
131
+ if (w.confidence !== undefined) tw.confidence = w.confidence
132
+ if (w.speaker !== undefined) tw.speaker = w.speaker
133
+ return tw
134
+ },
135
+ )
136
+
137
+ return {
138
+ id: generateId(this.name),
139
+ model,
140
+ text: data.text,
141
+ language: data.language ?? language,
142
+ duration: data.duration,
143
+ words,
144
+ }
145
+ } catch (error) {
146
+ logger.errors('grok.transcribe fatal', {
147
+ error,
148
+ source: 'grok.transcribe',
149
+ })
150
+ throw error
151
+ }
152
+ }
153
+ }
154
+
155
+ /**
156
+ * Build the multipart/form-data body for `POST /v1/stt`, coercing SDK-level
157
+ * model options into xAI's wire format (booleans as `'true'`/`'false'`
158
+ * strings, numeric fields stringified, etc.).
159
+ *
160
+ * Wire-field mapping:
161
+ * - `modelOptions.inverse_text_normalization` → `format` (xAI's chosen
162
+ * wire-field name for the ITN boolean; the SDK surfaces it under the
163
+ * clearer `inverse_text_normalization` key).
164
+ * - `modelOptions.audio_format`, `sample_rate`, `multichannel`, `channels`,
165
+ * `diarize` map to same-named form fields.
166
+ */
167
+ export function buildTranscriptionFormData(options: {
168
+ file: File
169
+ language: string | undefined
170
+ modelOptions: GrokTranscriptionProviderOptions | undefined
171
+ }): FormData {
172
+ const { file, language, modelOptions } = options
173
+ const form = new FormData()
174
+ form.set('file', file)
175
+ if (language) form.set('language', language)
176
+ if (modelOptions?.audio_format !== undefined) {
177
+ form.set('audio_format', modelOptions.audio_format)
178
+ }
179
+ if (modelOptions?.sample_rate !== undefined) {
180
+ form.set('sample_rate', String(modelOptions.sample_rate))
181
+ }
182
+ if (modelOptions?.inverse_text_normalization !== undefined) {
183
+ form.set(
184
+ 'format',
185
+ modelOptions.inverse_text_normalization ? 'true' : 'false',
186
+ )
187
+ }
188
+ if (modelOptions?.multichannel !== undefined) {
189
+ form.set('multichannel', modelOptions.multichannel ? 'true' : 'false')
190
+ }
191
+ if (modelOptions?.channels !== undefined) {
192
+ form.set('channels', String(modelOptions.channels))
193
+ }
194
+ if (modelOptions?.diarize !== undefined) {
195
+ form.set('diarize', modelOptions.diarize ? 'true' : 'false')
196
+ }
197
+ return form
198
+ }
199
+
200
+ /**
201
+ * Creates a Grok transcription adapter with an explicit API key.
202
+ *
203
+ * @example
204
+ * ```typescript
205
+ * const adapter = createGrokTranscription('grok-stt', 'xai-...')
206
+ * const result = await generateTranscription({
207
+ * adapter,
208
+ * audio: audioFile,
209
+ * language: 'en',
210
+ * })
211
+ * ```
212
+ */
213
+ export function createGrokTranscription<TModel extends GrokTranscriptionModel>(
214
+ model: TModel,
215
+ apiKey: string,
216
+ config?: Omit<GrokTranscriptionConfig, 'apiKey'>,
217
+ ): GrokTranscriptionAdapter<TModel> {
218
+ return new GrokTranscriptionAdapter({ apiKey, ...config }, model)
219
+ }
220
+
221
+ /**
222
+ * Creates a Grok transcription adapter, reading the API key from
223
+ * `XAI_API_KEY` in the environment.
224
+ *
225
+ * @throws Error if `XAI_API_KEY` is not set.
226
+ */
227
+ export function grokTranscription<TModel extends GrokTranscriptionModel>(
228
+ model: TModel,
229
+ config?: Omit<GrokTranscriptionConfig, 'apiKey'>,
230
+ ): GrokTranscriptionAdapter<TModel> {
231
+ const apiKey = getGrokApiKeyFromEnv()
232
+ return createGrokTranscription(model, apiKey, config)
233
+ }
@@ -0,0 +1,260 @@
1
+ import { BaseTTSAdapter } from '@tanstack/ai/adapters'
2
+ import { arrayBufferToBase64, generateId, getGrokApiKeyFromEnv } from '../utils'
3
+ import type { TTSOptions, TTSResult } from '@tanstack/ai'
4
+ import type { GrokTTSModel } from '../model-meta'
5
+ import type {
6
+ GrokTTSCodec,
7
+ GrokTTSProviderOptions,
8
+ GrokTTSVoice,
9
+ } from '../audio/tts-provider-options'
10
+
11
+ const DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'
12
+
13
+ /**
14
+ * Configuration for the Grok TTS adapter.
15
+ *
16
+ * Unlike chat/image/summarize adapters, TTS does not use the OpenAI SDK
17
+ * because xAI's `/v1/tts` endpoint is not OpenAI-compatible. This config
18
+ * is a minimal subset suitable for direct `fetch` calls.
19
+ */
20
+ export interface GrokSpeechConfig {
21
+ apiKey: string
22
+ baseURL?: string
23
+ /** Additional headers to merge into every request (e.g., test IDs). */
24
+ defaultHeaders?: Record<string, string>
25
+ }
26
+
27
+ /**
28
+ * Grok Text-to-Speech Adapter.
29
+ *
30
+ * Talks to `POST {baseURL}/tts` per
31
+ * https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
32
+ */
33
+ export class GrokSpeechAdapter<
34
+ TModel extends GrokTTSModel,
35
+ > extends BaseTTSAdapter<TModel, GrokTTSProviderOptions> {
36
+ readonly name = 'grok' as const
37
+
38
+ private readonly apiKey: string
39
+ private readonly baseURL: string
40
+ private readonly defaultHeaders: Record<string, string>
41
+
42
+ constructor(config: GrokSpeechConfig, model: TModel) {
43
+ super(model, config)
44
+ this.apiKey = config.apiKey
45
+ this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, '')
46
+ this.defaultHeaders = config.defaultHeaders ?? {}
47
+ }
48
+
49
+ async generateSpeech(
50
+ options: TTSOptions<GrokTTSProviderOptions>,
51
+ ): Promise<TTSResult> {
52
+ const { logger } = options
53
+ const { model, text, voice, format, modelOptions } = options
54
+
55
+ logger.request(`activity=generateSpeech provider=grok model=${model}`, {
56
+ provider: 'grok',
57
+ model,
58
+ })
59
+
60
+ const { body, codec, sampleRateForContentType } = buildTTSRequestBody({
61
+ text,
62
+ voice,
63
+ format,
64
+ modelOptions,
65
+ })
66
+
67
+ try {
68
+ const response = await fetch(`${this.baseURL}/tts`, {
69
+ method: 'POST',
70
+ headers: {
71
+ // `defaultHeaders` first so the adapter's Authorization / Content-Type
72
+ // always win — otherwise a caller-supplied `Authorization` header
73
+ // could silently clobber the bearer token.
74
+ ...this.defaultHeaders,
75
+ Authorization: `Bearer ${this.apiKey}`,
76
+ 'Content-Type': 'application/json',
77
+ },
78
+ body: JSON.stringify(body),
79
+ })
80
+
81
+ if (!response.ok) {
82
+ const errorText = await response.text()
83
+ throw new Error(
84
+ `Grok TTS request failed: ${response.status} ${errorText}`,
85
+ )
86
+ }
87
+
88
+ const arrayBuffer = await response.arrayBuffer()
89
+ const audio = arrayBufferToBase64(arrayBuffer)
90
+
91
+ return {
92
+ id: generateId(this.name),
93
+ model,
94
+ audio,
95
+ format: codec,
96
+ contentType: getContentType(codec, sampleRateForContentType),
97
+ }
98
+ } catch (error) {
99
+ logger.errors('grok.generateSpeech fatal', {
100
+ error,
101
+ source: 'grok.generateSpeech',
102
+ })
103
+ throw error
104
+ }
105
+ }
106
+ }
107
+
108
+ /**
109
+ * Build the JSON body for `POST /v1/tts`, resolving codec / sample-rate / voice
110
+ * defaults in one place.
111
+ *
112
+ * Returns the request `body`, the resolved `codec`, and the `sampleRateForContentType`
113
+ * used by the caller to label the response via `getContentType`.
114
+ */
115
+ export function buildTTSRequestBody(options: {
116
+ text: string
117
+ voice: string | undefined
118
+ format: TTSOptions['format'] | undefined
119
+ modelOptions: GrokTTSProviderOptions | undefined
120
+ }): {
121
+ body: Record<string, unknown>
122
+ codec: GrokTTSCodec
123
+ sampleRateForContentType: number
124
+ } {
125
+ const { text, voice, format, modelOptions } = options
126
+
127
+ const codec = pickCodec(modelOptions?.codec, format)
128
+
129
+ // Only forward `sample_rate` when either:
130
+ // - the caller explicitly set `modelOptions.sample_rate`, or
131
+ // - the codec's Content-Type carries the rate (pcm → audio/L16;rate=…).
132
+ // For mp3/wav/opus/aac/flac we leave sample_rate unset so xAI's server
133
+ // default applies.
134
+ const callerSampleRate = modelOptions?.sample_rate
135
+ // Default sample rate documented in GrokTTSProviderOptions is 24000 Hz —
136
+ // used only when we MUST attach a rate to the contentType (pcm) and the
137
+ // caller didn't pick one.
138
+ const pcmDefault = 24000
139
+ const needsRateInContentType = codec === 'pcm'
140
+
141
+ const outputFormat: Record<string, unknown> = { codec }
142
+ if (callerSampleRate !== undefined) {
143
+ outputFormat.sample_rate = callerSampleRate
144
+ } else if (needsRateInContentType) {
145
+ outputFormat.sample_rate = pcmDefault
146
+ }
147
+ if (codec === 'mp3' && modelOptions?.bit_rate !== undefined) {
148
+ outputFormat.bit_rate = modelOptions.bit_rate
149
+ }
150
+
151
+ // pcm embeds the rate in `audio/L16;rate=…`; mulaw/alaw embed it in
152
+ // `audio/PCMU;rate=…` / `audio/PCMA;rate=…` when non-default. mp3/wav
153
+ // don't carry a rate parameter so the value is unused for those.
154
+ const sampleRateForContentType = callerSampleRate ?? pcmDefault
155
+
156
+ const body: Record<string, unknown> = {
157
+ text,
158
+ voice_id: (voice as GrokTTSVoice | undefined) ?? 'eve',
159
+ language: modelOptions?.language ?? 'en',
160
+ output_format: outputFormat,
161
+ }
162
+ if (modelOptions?.optimize_streaming_latency !== undefined) {
163
+ body.optimize_streaming_latency = modelOptions.optimize_streaming_latency
164
+ }
165
+ if (modelOptions?.text_normalization !== undefined) {
166
+ body.text_normalization = modelOptions.text_normalization
167
+ }
168
+
169
+ return { body, codec, sampleRateForContentType }
170
+ }
171
+
172
+ /**
173
+ * Maps the cross-provider `TTSOptions.format` onto Grok's supported codecs.
174
+ * `opus`, `aac`, and `flac` are not supported by xAI TTS (which only exposes
175
+ * mp3/wav/pcm/mulaw/alaw) — we fall back to mp3. An explicit
176
+ * `modelOptions.codec` always wins.
177
+ */
178
+ function pickCodec(
179
+ codecOverride: GrokTTSCodec | undefined,
180
+ format: TTSOptions['format'] | undefined,
181
+ ): GrokTTSCodec {
182
+ if (codecOverride) return codecOverride
183
+ if (!format) return 'mp3'
184
+ switch (format) {
185
+ case 'mp3':
186
+ case 'wav':
187
+ case 'pcm':
188
+ return format
189
+ case 'flac':
190
+ case 'opus':
191
+ case 'aac':
192
+ return 'mp3'
193
+ default:
194
+ return 'mp3'
195
+ }
196
+ }
197
+
198
+ export function getContentType(
199
+ codec: GrokTTSCodec,
200
+ sampleRate: number,
201
+ ): string {
202
+ switch (codec) {
203
+ case 'mp3':
204
+ return 'audio/mpeg'
205
+ case 'wav':
206
+ return 'audio/wav'
207
+ case 'pcm':
208
+ // `audio/L16` requires a `rate` parameter per RFC 3551/3555.
209
+ return `audio/L16;rate=${sampleRate}`
210
+ case 'mulaw':
211
+ // `audio/basic` is 8 kHz mono by RFC 2046 registration. For non-8kHz
212
+ // streams xAI still produces mulaw-encoded bytes at the requested
213
+ // rate, but the registered MIME can't carry that rate — so we use
214
+ // the non-standard but commonly-supported `audio/PCMU;rate=…` (RFC 3551
215
+ // RTP payload name) whenever the caller asked for a rate other than
216
+ // 8000, and keep `audio/basic` for the standard 8kHz case.
217
+ return sampleRate === 8000
218
+ ? 'audio/basic'
219
+ : `audio/PCMU;rate=${sampleRate}`
220
+ case 'alaw':
221
+ return sampleRate === 8000
222
+ ? 'audio/x-alaw-basic'
223
+ : `audio/PCMA;rate=${sampleRate}`
224
+ }
225
+ }
226
+
227
+ /**
228
+ * Creates a Grok speech (TTS) adapter with an explicit API key.
229
+ *
230
+ * @example
231
+ * ```typescript
232
+ * const adapter = createGrokSpeech('grok-tts', 'xai-...')
233
+ * const result = await generateSpeech({
234
+ * adapter,
235
+ * text: 'Hello from Grok',
236
+ * voice: 'eve',
237
+ * })
238
+ * ```
239
+ */
240
+ export function createGrokSpeech<TModel extends GrokTTSModel>(
241
+ model: TModel,
242
+ apiKey: string,
243
+ config?: Omit<GrokSpeechConfig, 'apiKey'>,
244
+ ): GrokSpeechAdapter<TModel> {
245
+ return new GrokSpeechAdapter({ apiKey, ...config }, model)
246
+ }
247
+
248
+ /**
249
+ * Creates a Grok speech (TTS) adapter, reading the API key from
250
+ * `XAI_API_KEY` in the environment.
251
+ *
252
+ * @throws Error if `XAI_API_KEY` is not set.
253
+ */
254
+ export function grokSpeech<TModel extends GrokTTSModel>(
255
+ model: TModel,
256
+ config?: Omit<GrokSpeechConfig, 'apiKey'>,
257
+ ): GrokSpeechAdapter<TModel> {
258
+ const apiKey = getGrokApiKeyFromEnv()
259
+ return createGrokSpeech(model, apiKey, config)
260
+ }
@@ -0,0 +1,54 @@
1
+ /**
2
+ * Grok STT supported audio formats.
3
+ * See https://docs.x.ai/developers/rest-api-reference/inference/voice
4
+ */
5
+ export type GrokSTTAudioFormat =
6
+ | 'pcm'
7
+ | 'mulaw'
8
+ | 'alaw'
9
+ | 'wav'
10
+ | 'mp3'
11
+ | 'ogg'
12
+ | 'opus'
13
+ | 'flac'
14
+ | 'aac'
15
+ | 'mp4'
16
+ | 'm4a'
17
+ | 'mkv'
18
+
19
+ /**
20
+ * Provider-specific options for Grok transcription (`POST /v1/stt`).
21
+ */
22
+ export interface GrokTranscriptionProviderOptions {
23
+ /**
24
+ * The format of the provided audio. Required for raw codecs (pcm, mulaw, alaw).
25
+ */
26
+ audio_format?: GrokSTTAudioFormat
27
+ /**
28
+ * Sample rate of the audio (Hz). Required for raw codecs.
29
+ */
30
+ sample_rate?: number
31
+ /**
32
+ * Apply inverse text normalization (e.g. "one hundred" → "100"). Requires
33
+ * `language` to be set on the core `TranscriptionOptions`.
34
+ *
35
+ * NOTE: xAI's STT API exposes this on the wire as `format` (a boolean
36
+ * toggle). We surface it under the clearer name
37
+ * `inverse_text_normalization` on the SDK, and translate to the wire name
38
+ * inside the adapter.
39
+ */
40
+ inverse_text_normalization?: boolean
41
+ /**
42
+ * Treat the audio as multichannel. When enabled, `channels` must also be set.
43
+ */
44
+ multichannel?: boolean
45
+ /**
46
+ * Channel count for multichannel raw audio (2–8).
47
+ */
48
+ channels?: number
49
+ /**
50
+ * Enable speaker diarization. When true, response words include a `speaker`
51
+ * field.
52
+ */
53
+ diarize?: boolean
54
+ }
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Grok TTS voice options.
3
+ * See https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
4
+ */
5
+ export type GrokTTSVoice = 'eve' | 'ara' | 'rex' | 'sal' | 'leo'
6
+
7
+ /**
8
+ * Grok TTS output audio codecs.
9
+ * Grok does NOT support opus or aac; those formats are mapped to mp3.
10
+ */
11
+ export type GrokTTSCodec = 'mp3' | 'wav' | 'pcm' | 'mulaw' | 'alaw'
12
+
13
+ /**
14
+ * Provider-specific options for Grok TTS (`POST /v1/tts`).
15
+ */
16
+ export interface GrokTTSProviderOptions {
17
+ /**
18
+ * BCP-47 language code (e.g., `en`, `zh`, `pt-BR`) or `'auto'` for detection.
19
+ * Defaults to `'en'` when not provided.
20
+ */
21
+ language?: string
22
+ /**
23
+ * Audio codec. Overrides the `format` field on `TTSOptions` when set.
24
+ */
25
+ codec?: GrokTTSCodec
26
+ /**
27
+ * Sample rate in Hz. Valid values: 8000, 16000, 22050, 24000, 44100, 48000.
28
+ * Defaults to 24000.
29
+ */
30
+ sample_rate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000
31
+ /**
32
+ * Bit rate for MP3 output. Ignored for other codecs.
33
+ * Valid values: 32000, 64000, 96000, 128000, 192000. Defaults to 128000.
34
+ */
35
+ bit_rate?: 32000 | 64000 | 96000 | 128000 | 192000
36
+ /**
37
+ * Set to 1 for lower latency streaming; 0 (default) for normal quality.
38
+ */
39
+ optimize_streaming_latency?: 0 | 1
40
+ /**
41
+ * Enable text normalization. Defaults to false.
42
+ */
43
+ text_normalization?: boolean
44
+ }