@tanstack/ai-grok 0.6.7 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.js +36 -17
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/summarize.js +51 -22
- package/dist/esm/adapters/summarize.js.map +1 -1
- package/dist/esm/adapters/text.js +25 -10
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/transcription.d.ts +84 -0
- package/dist/esm/adapters/transcription.js +109 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +70 -0
- package/dist/esm/adapters/tts.js +137 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +41 -0
- package/dist/esm/audio/tts-provider-options.d.ts +42 -0
- package/dist/esm/index.d.ts +8 -2
- package/dist/esm/index.js +17 -2
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +6 -0
- package/dist/esm/model-meta.js +22 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.d.ts +21 -0
- package/dist/esm/realtime/adapter.js +816 -0
- package/dist/esm/realtime/adapter.js.map +1 -0
- package/dist/esm/realtime/index.d.ts +4 -0
- package/dist/esm/realtime/realtime-contract.d.ts +30 -0
- package/dist/esm/realtime/token.d.ts +22 -0
- package/dist/esm/realtime/token.js +73 -0
- package/dist/esm/realtime/token.js.map +1 -0
- package/dist/esm/realtime/types.d.ts +95 -0
- package/dist/esm/utils/audio.d.ts +23 -0
- package/dist/esm/utils/audio.js +171 -0
- package/dist/esm/utils/audio.js.map +1 -0
- package/dist/esm/utils/index.d.ts +1 -0
- package/package.json +6 -3
- package/src/adapters/image.ts +41 -19
- package/src/adapters/summarize.ts +56 -25
- package/src/adapters/text.ts +26 -9
- package/src/adapters/transcription.ts +233 -0
- package/src/adapters/tts.ts +260 -0
- package/src/audio/transcription-provider-options.ts +54 -0
- package/src/audio/tts-provider-options.ts +44 -0
- package/src/index.ts +50 -1
- package/src/model-meta.ts +54 -0
- package/src/realtime/adapter.ts +1215 -0
- package/src/realtime/index.ts +18 -0
- package/src/realtime/realtime-contract.ts +46 -0
- package/src/realtime/token.ts +131 -0
- package/src/realtime/types.ts +105 -0
- package/src/utils/audio.ts +217 -0
- package/src/utils/index.ts +1 -0
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
import { BaseTTSAdapter } from '@tanstack/ai/adapters'
|
|
2
|
+
import { arrayBufferToBase64, generateId, getGrokApiKeyFromEnv } from '../utils'
|
|
3
|
+
import type { TTSOptions, TTSResult } from '@tanstack/ai'
|
|
4
|
+
import type { GrokTTSModel } from '../model-meta'
|
|
5
|
+
import type {
|
|
6
|
+
GrokTTSCodec,
|
|
7
|
+
GrokTTSProviderOptions,
|
|
8
|
+
GrokTTSVoice,
|
|
9
|
+
} from '../audio/tts-provider-options'
|
|
10
|
+
|
|
11
|
+
const DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Configuration for the Grok TTS adapter.
|
|
15
|
+
*
|
|
16
|
+
* Unlike chat/image/summarize adapters, TTS does not use the OpenAI SDK
|
|
17
|
+
* because xAI's `/v1/tts` endpoint is not OpenAI-compatible. This config
|
|
18
|
+
* is a minimal subset suitable for direct `fetch` calls.
|
|
19
|
+
*/
|
|
20
|
+
export interface GrokSpeechConfig {
|
|
21
|
+
apiKey: string
|
|
22
|
+
baseURL?: string
|
|
23
|
+
/** Additional headers to merge into every request (e.g., test IDs). */
|
|
24
|
+
defaultHeaders?: Record<string, string>
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Grok Text-to-Speech Adapter.
|
|
29
|
+
*
|
|
30
|
+
* Talks to `POST {baseURL}/tts` per
|
|
31
|
+
* https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
|
|
32
|
+
*/
|
|
33
|
+
export class GrokSpeechAdapter<
|
|
34
|
+
TModel extends GrokTTSModel,
|
|
35
|
+
> extends BaseTTSAdapter<TModel, GrokTTSProviderOptions> {
|
|
36
|
+
readonly name = 'grok' as const
|
|
37
|
+
|
|
38
|
+
private readonly apiKey: string
|
|
39
|
+
private readonly baseURL: string
|
|
40
|
+
private readonly defaultHeaders: Record<string, string>
|
|
41
|
+
|
|
42
|
+
constructor(config: GrokSpeechConfig, model: TModel) {
|
|
43
|
+
super(model, config)
|
|
44
|
+
this.apiKey = config.apiKey
|
|
45
|
+
this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, '')
|
|
46
|
+
this.defaultHeaders = config.defaultHeaders ?? {}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
async generateSpeech(
|
|
50
|
+
options: TTSOptions<GrokTTSProviderOptions>,
|
|
51
|
+
): Promise<TTSResult> {
|
|
52
|
+
const { logger } = options
|
|
53
|
+
const { model, text, voice, format, modelOptions } = options
|
|
54
|
+
|
|
55
|
+
logger.request(`activity=generateSpeech provider=grok model=${model}`, {
|
|
56
|
+
provider: 'grok',
|
|
57
|
+
model,
|
|
58
|
+
})
|
|
59
|
+
|
|
60
|
+
const { body, codec, sampleRateForContentType } = buildTTSRequestBody({
|
|
61
|
+
text,
|
|
62
|
+
voice,
|
|
63
|
+
format,
|
|
64
|
+
modelOptions,
|
|
65
|
+
})
|
|
66
|
+
|
|
67
|
+
try {
|
|
68
|
+
const response = await fetch(`${this.baseURL}/tts`, {
|
|
69
|
+
method: 'POST',
|
|
70
|
+
headers: {
|
|
71
|
+
// `defaultHeaders` first so the adapter's Authorization / Content-Type
|
|
72
|
+
// always win — otherwise a caller-supplied `Authorization` header
|
|
73
|
+
// could silently clobber the bearer token.
|
|
74
|
+
...this.defaultHeaders,
|
|
75
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
76
|
+
'Content-Type': 'application/json',
|
|
77
|
+
},
|
|
78
|
+
body: JSON.stringify(body),
|
|
79
|
+
})
|
|
80
|
+
|
|
81
|
+
if (!response.ok) {
|
|
82
|
+
const errorText = await response.text()
|
|
83
|
+
throw new Error(
|
|
84
|
+
`Grok TTS request failed: ${response.status} ${errorText}`,
|
|
85
|
+
)
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
const arrayBuffer = await response.arrayBuffer()
|
|
89
|
+
const audio = arrayBufferToBase64(arrayBuffer)
|
|
90
|
+
|
|
91
|
+
return {
|
|
92
|
+
id: generateId(this.name),
|
|
93
|
+
model,
|
|
94
|
+
audio,
|
|
95
|
+
format: codec,
|
|
96
|
+
contentType: getContentType(codec, sampleRateForContentType),
|
|
97
|
+
}
|
|
98
|
+
} catch (error) {
|
|
99
|
+
logger.errors('grok.generateSpeech fatal', {
|
|
100
|
+
error,
|
|
101
|
+
source: 'grok.generateSpeech',
|
|
102
|
+
})
|
|
103
|
+
throw error
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Build the JSON body for `POST /v1/tts`, resolving codec / sample-rate / voice
|
|
110
|
+
* defaults in one place.
|
|
111
|
+
*
|
|
112
|
+
* Returns the request `body`, the resolved `codec`, and the `sampleRateForContentType`
|
|
113
|
+
* used by the caller to label the response via `getContentType`.
|
|
114
|
+
*/
|
|
115
|
+
export function buildTTSRequestBody(options: {
|
|
116
|
+
text: string
|
|
117
|
+
voice: string | undefined
|
|
118
|
+
format: TTSOptions['format'] | undefined
|
|
119
|
+
modelOptions: GrokTTSProviderOptions | undefined
|
|
120
|
+
}): {
|
|
121
|
+
body: Record<string, unknown>
|
|
122
|
+
codec: GrokTTSCodec
|
|
123
|
+
sampleRateForContentType: number
|
|
124
|
+
} {
|
|
125
|
+
const { text, voice, format, modelOptions } = options
|
|
126
|
+
|
|
127
|
+
const codec = pickCodec(modelOptions?.codec, format)
|
|
128
|
+
|
|
129
|
+
// Only forward `sample_rate` when either:
|
|
130
|
+
// - the caller explicitly set `modelOptions.sample_rate`, or
|
|
131
|
+
// - the codec's Content-Type carries the rate (pcm → audio/L16;rate=…).
|
|
132
|
+
// For mp3/wav/opus/aac/flac we leave sample_rate unset so xAI's server
|
|
133
|
+
// default applies.
|
|
134
|
+
const callerSampleRate = modelOptions?.sample_rate
|
|
135
|
+
// Default sample rate documented in GrokTTSProviderOptions is 24000 Hz —
|
|
136
|
+
// used only when we MUST attach a rate to the contentType (pcm) and the
|
|
137
|
+
// caller didn't pick one.
|
|
138
|
+
const pcmDefault = 24000
|
|
139
|
+
const needsRateInContentType = codec === 'pcm'
|
|
140
|
+
|
|
141
|
+
const outputFormat: Record<string, unknown> = { codec }
|
|
142
|
+
if (callerSampleRate !== undefined) {
|
|
143
|
+
outputFormat.sample_rate = callerSampleRate
|
|
144
|
+
} else if (needsRateInContentType) {
|
|
145
|
+
outputFormat.sample_rate = pcmDefault
|
|
146
|
+
}
|
|
147
|
+
if (codec === 'mp3' && modelOptions?.bit_rate !== undefined) {
|
|
148
|
+
outputFormat.bit_rate = modelOptions.bit_rate
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
// pcm embeds the rate in `audio/L16;rate=…`; mulaw/alaw embed it in
|
|
152
|
+
// `audio/PCMU;rate=…` / `audio/PCMA;rate=…` when non-default. mp3/wav
|
|
153
|
+
// don't carry a rate parameter so the value is unused for those.
|
|
154
|
+
const sampleRateForContentType = callerSampleRate ?? pcmDefault
|
|
155
|
+
|
|
156
|
+
const body: Record<string, unknown> = {
|
|
157
|
+
text,
|
|
158
|
+
voice_id: (voice as GrokTTSVoice | undefined) ?? 'eve',
|
|
159
|
+
language: modelOptions?.language ?? 'en',
|
|
160
|
+
output_format: outputFormat,
|
|
161
|
+
}
|
|
162
|
+
if (modelOptions?.optimize_streaming_latency !== undefined) {
|
|
163
|
+
body.optimize_streaming_latency = modelOptions.optimize_streaming_latency
|
|
164
|
+
}
|
|
165
|
+
if (modelOptions?.text_normalization !== undefined) {
|
|
166
|
+
body.text_normalization = modelOptions.text_normalization
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
return { body, codec, sampleRateForContentType }
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Maps the cross-provider `TTSOptions.format` onto Grok's supported codecs.
|
|
174
|
+
* `opus`, `aac`, and `flac` are not supported by xAI TTS (which only exposes
|
|
175
|
+
* mp3/wav/pcm/mulaw/alaw) — we fall back to mp3. An explicit
|
|
176
|
+
* `modelOptions.codec` always wins.
|
|
177
|
+
*/
|
|
178
|
+
function pickCodec(
|
|
179
|
+
codecOverride: GrokTTSCodec | undefined,
|
|
180
|
+
format: TTSOptions['format'] | undefined,
|
|
181
|
+
): GrokTTSCodec {
|
|
182
|
+
if (codecOverride) return codecOverride
|
|
183
|
+
if (!format) return 'mp3'
|
|
184
|
+
switch (format) {
|
|
185
|
+
case 'mp3':
|
|
186
|
+
case 'wav':
|
|
187
|
+
case 'pcm':
|
|
188
|
+
return format
|
|
189
|
+
case 'flac':
|
|
190
|
+
case 'opus':
|
|
191
|
+
case 'aac':
|
|
192
|
+
return 'mp3'
|
|
193
|
+
default:
|
|
194
|
+
return 'mp3'
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
export function getContentType(
|
|
199
|
+
codec: GrokTTSCodec,
|
|
200
|
+
sampleRate: number,
|
|
201
|
+
): string {
|
|
202
|
+
switch (codec) {
|
|
203
|
+
case 'mp3':
|
|
204
|
+
return 'audio/mpeg'
|
|
205
|
+
case 'wav':
|
|
206
|
+
return 'audio/wav'
|
|
207
|
+
case 'pcm':
|
|
208
|
+
// `audio/L16` requires a `rate` parameter per RFC 3551/3555.
|
|
209
|
+
return `audio/L16;rate=${sampleRate}`
|
|
210
|
+
case 'mulaw':
|
|
211
|
+
// `audio/basic` is 8 kHz mono by RFC 2046 registration. For non-8kHz
|
|
212
|
+
// streams xAI still produces mulaw-encoded bytes at the requested
|
|
213
|
+
// rate, but the registered MIME can't carry that rate — so we use
|
|
214
|
+
// the non-standard but commonly-supported `audio/PCMU;rate=…` (RFC 3551
|
|
215
|
+
// RTP payload name) whenever the caller asked for a rate other than
|
|
216
|
+
// 8000, and keep `audio/basic` for the standard 8kHz case.
|
|
217
|
+
return sampleRate === 8000
|
|
218
|
+
? 'audio/basic'
|
|
219
|
+
: `audio/PCMU;rate=${sampleRate}`
|
|
220
|
+
case 'alaw':
|
|
221
|
+
return sampleRate === 8000
|
|
222
|
+
? 'audio/x-alaw-basic'
|
|
223
|
+
: `audio/PCMA;rate=${sampleRate}`
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* Creates a Grok speech (TTS) adapter with an explicit API key.
|
|
229
|
+
*
|
|
230
|
+
* @example
|
|
231
|
+
* ```typescript
|
|
232
|
+
* const adapter = createGrokSpeech('grok-tts', 'xai-...')
|
|
233
|
+
* const result = await generateSpeech({
|
|
234
|
+
* adapter,
|
|
235
|
+
* text: 'Hello from Grok',
|
|
236
|
+
* voice: 'eve',
|
|
237
|
+
* })
|
|
238
|
+
* ```
|
|
239
|
+
*/
|
|
240
|
+
export function createGrokSpeech<TModel extends GrokTTSModel>(
|
|
241
|
+
model: TModel,
|
|
242
|
+
apiKey: string,
|
|
243
|
+
config?: Omit<GrokSpeechConfig, 'apiKey'>,
|
|
244
|
+
): GrokSpeechAdapter<TModel> {
|
|
245
|
+
return new GrokSpeechAdapter({ apiKey, ...config }, model)
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Creates a Grok speech (TTS) adapter, reading the API key from
|
|
250
|
+
* `XAI_API_KEY` in the environment.
|
|
251
|
+
*
|
|
252
|
+
* @throws Error if `XAI_API_KEY` is not set.
|
|
253
|
+
*/
|
|
254
|
+
export function grokSpeech<TModel extends GrokTTSModel>(
|
|
255
|
+
model: TModel,
|
|
256
|
+
config?: Omit<GrokSpeechConfig, 'apiKey'>,
|
|
257
|
+
): GrokSpeechAdapter<TModel> {
|
|
258
|
+
const apiKey = getGrokApiKeyFromEnv()
|
|
259
|
+
return createGrokSpeech(model, apiKey, config)
|
|
260
|
+
}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Grok STT supported audio formats.
|
|
3
|
+
* See https://docs.x.ai/developers/rest-api-reference/inference/voice
|
|
4
|
+
*/
|
|
5
|
+
export type GrokSTTAudioFormat =
|
|
6
|
+
| 'pcm'
|
|
7
|
+
| 'mulaw'
|
|
8
|
+
| 'alaw'
|
|
9
|
+
| 'wav'
|
|
10
|
+
| 'mp3'
|
|
11
|
+
| 'ogg'
|
|
12
|
+
| 'opus'
|
|
13
|
+
| 'flac'
|
|
14
|
+
| 'aac'
|
|
15
|
+
| 'mp4'
|
|
16
|
+
| 'm4a'
|
|
17
|
+
| 'mkv'
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Provider-specific options for Grok transcription (`POST /v1/stt`).
|
|
21
|
+
*/
|
|
22
|
+
export interface GrokTranscriptionProviderOptions {
|
|
23
|
+
/**
|
|
24
|
+
* The format of the provided audio. Required for raw codecs (pcm, mulaw, alaw).
|
|
25
|
+
*/
|
|
26
|
+
audio_format?: GrokSTTAudioFormat
|
|
27
|
+
/**
|
|
28
|
+
* Sample rate of the audio (Hz). Required for raw codecs.
|
|
29
|
+
*/
|
|
30
|
+
sample_rate?: number
|
|
31
|
+
/**
|
|
32
|
+
* Apply inverse text normalization (e.g. "one hundred" → "100"). Requires
|
|
33
|
+
* `language` to be set on the core `TranscriptionOptions`.
|
|
34
|
+
*
|
|
35
|
+
* NOTE: xAI's STT API exposes this on the wire as `format` (a boolean
|
|
36
|
+
* toggle). We surface it under the clearer name
|
|
37
|
+
* `inverse_text_normalization` on the SDK, and translate to the wire name
|
|
38
|
+
* inside the adapter.
|
|
39
|
+
*/
|
|
40
|
+
inverse_text_normalization?: boolean
|
|
41
|
+
/**
|
|
42
|
+
* Treat the audio as multichannel. When enabled, `channels` must also be set.
|
|
43
|
+
*/
|
|
44
|
+
multichannel?: boolean
|
|
45
|
+
/**
|
|
46
|
+
* Channel count for multichannel raw audio (2–8).
|
|
47
|
+
*/
|
|
48
|
+
channels?: number
|
|
49
|
+
/**
|
|
50
|
+
* Enable speaker diarization. When true, response words include a `speaker`
|
|
51
|
+
* field.
|
|
52
|
+
*/
|
|
53
|
+
diarize?: boolean
|
|
54
|
+
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Grok TTS voice options.
|
|
3
|
+
* See https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
|
|
4
|
+
*/
|
|
5
|
+
export type GrokTTSVoice = 'eve' | 'ara' | 'rex' | 'sal' | 'leo'
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Grok TTS output audio codecs.
|
|
9
|
+
* Grok does NOT support opus or aac; those formats are mapped to mp3.
|
|
10
|
+
*/
|
|
11
|
+
export type GrokTTSCodec = 'mp3' | 'wav' | 'pcm' | 'mulaw' | 'alaw'
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Provider-specific options for Grok TTS (`POST /v1/tts`).
|
|
15
|
+
*/
|
|
16
|
+
export interface GrokTTSProviderOptions {
|
|
17
|
+
/**
|
|
18
|
+
* BCP-47 language code (e.g., `en`, `zh`, `pt-BR`) or `'auto'` for detection.
|
|
19
|
+
* Defaults to `'en'` when not provided.
|
|
20
|
+
*/
|
|
21
|
+
language?: string
|
|
22
|
+
/**
|
|
23
|
+
* Audio codec. Overrides the `format` field on `TTSOptions` when set.
|
|
24
|
+
*/
|
|
25
|
+
codec?: GrokTTSCodec
|
|
26
|
+
/**
|
|
27
|
+
* Sample rate in Hz. Valid values: 8000, 16000, 22050, 24000, 44100, 48000.
|
|
28
|
+
* Defaults to 24000.
|
|
29
|
+
*/
|
|
30
|
+
sample_rate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000
|
|
31
|
+
/**
|
|
32
|
+
* Bit rate for MP3 output. Ignored for other codecs.
|
|
33
|
+
* Valid values: 32000, 64000, 96000, 128000, 192000. Defaults to 128000.
|
|
34
|
+
*/
|
|
35
|
+
bit_rate?: 32000 | 64000 | 96000 | 128000 | 192000
|
|
36
|
+
/**
|
|
37
|
+
* Set to 1 for lower latency streaming; 0 (default) for normal quality.
|
|
38
|
+
*/
|
|
39
|
+
optimize_streaming_latency?: 0 | 1
|
|
40
|
+
/**
|
|
41
|
+
* Enable text normalization. Defaults to false.
|
|
42
|
+
*/
|
|
43
|
+
text_normalization?: boolean
|
|
44
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -33,6 +33,31 @@ export type {
|
|
|
33
33
|
GrokImageModelProviderOptionsByName,
|
|
34
34
|
} from './image/image-provider-options'
|
|
35
35
|
|
|
36
|
+
// Speech (TTS) adapter - for text-to-speech
|
|
37
|
+
export {
|
|
38
|
+
GrokSpeechAdapter,
|
|
39
|
+
createGrokSpeech,
|
|
40
|
+
grokSpeech,
|
|
41
|
+
type GrokSpeechConfig,
|
|
42
|
+
} from './adapters/tts'
|
|
43
|
+
export type {
|
|
44
|
+
GrokTTSProviderOptions,
|
|
45
|
+
GrokTTSVoice,
|
|
46
|
+
GrokTTSCodec,
|
|
47
|
+
} from './audio/tts-provider-options'
|
|
48
|
+
|
|
49
|
+
// Transcription adapter - for speech-to-text
|
|
50
|
+
export {
|
|
51
|
+
GrokTranscriptionAdapter,
|
|
52
|
+
createGrokTranscription,
|
|
53
|
+
grokTranscription,
|
|
54
|
+
type GrokTranscriptionConfig,
|
|
55
|
+
} from './adapters/transcription'
|
|
56
|
+
export type {
|
|
57
|
+
GrokTranscriptionProviderOptions,
|
|
58
|
+
GrokSTTAudioFormat,
|
|
59
|
+
} from './audio/transcription-provider-options'
|
|
60
|
+
|
|
36
61
|
// ============================================================================
|
|
37
62
|
// Type Exports
|
|
38
63
|
// ============================================================================
|
|
@@ -45,8 +70,17 @@ export type {
|
|
|
45
70
|
ResolveInputModalities,
|
|
46
71
|
GrokChatModel,
|
|
47
72
|
GrokImageModel,
|
|
73
|
+
GrokTTSModel,
|
|
74
|
+
GrokTranscriptionModel,
|
|
75
|
+
GrokRealtimeModel,
|
|
76
|
+
} from './model-meta'
|
|
77
|
+
export {
|
|
78
|
+
GROK_CHAT_MODELS,
|
|
79
|
+
GROK_IMAGE_MODELS,
|
|
80
|
+
GROK_TTS_MODELS,
|
|
81
|
+
GROK_TRANSCRIPTION_MODELS,
|
|
82
|
+
GROK_REALTIME_MODELS,
|
|
48
83
|
} from './model-meta'
|
|
49
|
-
export { GROK_CHAT_MODELS, GROK_IMAGE_MODELS } from './model-meta'
|
|
50
84
|
export type {
|
|
51
85
|
GrokTextMetadata,
|
|
52
86
|
GrokImageMetadata,
|
|
@@ -55,3 +89,18 @@ export type {
|
|
|
55
89
|
GrokDocumentMetadata,
|
|
56
90
|
GrokMessageMetadataByModality,
|
|
57
91
|
} from './message-types'
|
|
92
|
+
|
|
93
|
+
// ============================================================================
|
|
94
|
+
// Realtime (Voice Agent) Adapters
|
|
95
|
+
// ============================================================================
|
|
96
|
+
|
|
97
|
+
export { grokRealtimeToken, grokRealtime } from './realtime/index'
|
|
98
|
+
|
|
99
|
+
export type {
|
|
100
|
+
GrokRealtimeVoice,
|
|
101
|
+
GrokRealtimeTokenOptions,
|
|
102
|
+
GrokRealtimeOptions,
|
|
103
|
+
GrokTurnDetection,
|
|
104
|
+
GrokSemanticVADConfig,
|
|
105
|
+
GrokServerVADConfig,
|
|
106
|
+
} from './realtime/index'
|
package/src/model-meta.ts
CHANGED
|
@@ -283,8 +283,62 @@ export const GROK_CHAT_MODELS = [
|
|
|
283
283
|
*/
|
|
284
284
|
export const GROK_IMAGE_MODELS = [GROK_2_IMAGE.name] as const
|
|
285
285
|
|
|
286
|
+
// xAI's `/v1/tts` endpoint is endpoint-addressed and does not take a `model`
|
|
287
|
+
// parameter. This synthetic identifier satisfies the SDK's `TTSOptions.model`
|
|
288
|
+
// contract and provides a stable value for logging and fixture matching.
|
|
289
|
+
const GROK_TTS = {
|
|
290
|
+
name: 'grok-tts',
|
|
291
|
+
supports: {
|
|
292
|
+
input: ['text'],
|
|
293
|
+
output: ['audio'],
|
|
294
|
+
},
|
|
295
|
+
} as const satisfies ModelMeta
|
|
296
|
+
|
|
297
|
+
// xAI's `/v1/stt` endpoint is endpoint-addressed and does not take a `model`
|
|
298
|
+
// parameter. This synthetic identifier satisfies the SDK's
|
|
299
|
+
// `TranscriptionOptions.model` contract.
|
|
300
|
+
const GROK_STT = {
|
|
301
|
+
name: 'grok-stt',
|
|
302
|
+
supports: {
|
|
303
|
+
input: ['audio'],
|
|
304
|
+
output: ['text'],
|
|
305
|
+
},
|
|
306
|
+
} as const satisfies ModelMeta
|
|
307
|
+
|
|
308
|
+
const GROK_VOICE_FAST_1 = {
|
|
309
|
+
name: 'grok-voice-fast-1.0',
|
|
310
|
+
supports: {
|
|
311
|
+
input: ['audio', 'text'],
|
|
312
|
+
output: ['audio', 'text'],
|
|
313
|
+
capabilities: ['tool_calling'],
|
|
314
|
+
tools: [] as const,
|
|
315
|
+
},
|
|
316
|
+
} as const satisfies ModelMeta
|
|
317
|
+
|
|
318
|
+
const GROK_VOICE_THINK_FAST_1 = {
|
|
319
|
+
name: 'grok-voice-think-fast-1.0',
|
|
320
|
+
supports: {
|
|
321
|
+
input: ['audio', 'text'],
|
|
322
|
+
output: ['audio', 'text'],
|
|
323
|
+
capabilities: ['reasoning', 'tool_calling'],
|
|
324
|
+
tools: [] as const,
|
|
325
|
+
},
|
|
326
|
+
} as const satisfies ModelMeta
|
|
327
|
+
|
|
328
|
+
export const GROK_TTS_MODELS = [GROK_TTS.name] as const
|
|
329
|
+
|
|
330
|
+
export const GROK_TRANSCRIPTION_MODELS = [GROK_STT.name] as const
|
|
331
|
+
|
|
332
|
+
export const GROK_REALTIME_MODELS = [
|
|
333
|
+
GROK_VOICE_FAST_1.name,
|
|
334
|
+
GROK_VOICE_THINK_FAST_1.name,
|
|
335
|
+
] as const
|
|
336
|
+
|
|
286
337
|
export type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]
|
|
287
338
|
export type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]
|
|
339
|
+
export type GrokTTSModel = (typeof GROK_TTS_MODELS)[number]
|
|
340
|
+
export type GrokTranscriptionModel = (typeof GROK_TRANSCRIPTION_MODELS)[number]
|
|
341
|
+
export type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]
|
|
288
342
|
|
|
289
343
|
/**
|
|
290
344
|
* Type-only map from Grok chat model name to its supported input modalities.
|