@tanstack/ai-grok 0.6.8 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.js +15 -8
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/transcription.d.ts +84 -0
- package/dist/esm/adapters/transcription.js +109 -0
- package/dist/esm/adapters/transcription.js.map +1 -0
- package/dist/esm/adapters/tts.d.ts +70 -0
- package/dist/esm/adapters/tts.js +137 -0
- package/dist/esm/adapters/tts.js.map +1 -0
- package/dist/esm/audio/transcription-provider-options.d.ts +41 -0
- package/dist/esm/audio/tts-provider-options.d.ts +42 -0
- package/dist/esm/index.d.ts +8 -2
- package/dist/esm/index.js +17 -2
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +6 -0
- package/dist/esm/model-meta.js +22 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.d.ts +21 -0
- package/dist/esm/realtime/adapter.js +816 -0
- package/dist/esm/realtime/adapter.js.map +1 -0
- package/dist/esm/realtime/index.d.ts +4 -0
- package/dist/esm/realtime/realtime-contract.d.ts +30 -0
- package/dist/esm/realtime/token.d.ts +22 -0
- package/dist/esm/realtime/token.js +73 -0
- package/dist/esm/realtime/token.js.map +1 -0
- package/dist/esm/realtime/types.d.ts +95 -0
- package/dist/esm/utils/audio.d.ts +23 -0
- package/dist/esm/utils/audio.js +171 -0
- package/dist/esm/utils/audio.js.map +1 -0
- package/dist/esm/utils/index.d.ts +1 -0
- package/package.json +6 -3
- package/src/adapters/image.ts +16 -7
- package/src/adapters/transcription.ts +233 -0
- package/src/adapters/tts.ts +260 -0
- package/src/audio/transcription-provider-options.ts +54 -0
- package/src/audio/tts-provider-options.ts +44 -0
- package/src/index.ts +50 -1
- package/src/model-meta.ts +54 -0
- package/src/realtime/adapter.ts +1215 -0
- package/src/realtime/index.ts +18 -0
- package/src/realtime/realtime-contract.ts +46 -0
- package/src/realtime/token.ts +131 -0
- package/src/realtime/types.ts +105 -0
- package/src/utils/audio.ts +217 -0
- package/src/utils/index.ts +1 -0
package/src/adapters/image.ts
CHANGED
|
@@ -49,7 +49,7 @@ export class GrokImageAdapter<
|
|
|
49
49
|
private client: OpenAI_SDK
|
|
50
50
|
|
|
51
51
|
constructor(config: GrokImageConfig, model: TModel) {
|
|
52
|
-
super({}
|
|
52
|
+
super(model, {})
|
|
53
53
|
this.client = createGrokClient(config)
|
|
54
54
|
}
|
|
55
55
|
|
|
@@ -92,12 +92,14 @@ export class GrokImageAdapter<
|
|
|
92
92
|
): OpenAI_SDK.Images.ImageGenerateParams {
|
|
93
93
|
const { model, prompt, numberOfImages, size, modelOptions } = options
|
|
94
94
|
|
|
95
|
+
// Spread modelOptions FIRST so explicit args (model, prompt, n, size) win
|
|
96
|
+
// and user-supplied modelOptions cannot silently override them.
|
|
95
97
|
return {
|
|
98
|
+
...modelOptions,
|
|
96
99
|
model,
|
|
97
100
|
prompt,
|
|
98
101
|
n: numberOfImages ?? 1,
|
|
99
102
|
size: size as OpenAI_SDK.Images.ImageGenerateParams['size'],
|
|
100
|
-
...modelOptions,
|
|
101
103
|
}
|
|
102
104
|
}
|
|
103
105
|
|
|
@@ -105,11 +107,18 @@ export class GrokImageAdapter<
|
|
|
105
107
|
model: string,
|
|
106
108
|
response: OpenAI_SDK.Images.ImagesResponse,
|
|
107
109
|
): ImageGenerationResult {
|
|
108
|
-
const images: Array<GeneratedImage> = (response.data ?? []).
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
110
|
+
const images: Array<GeneratedImage> = (response.data ?? []).flatMap(
|
|
111
|
+
(item): Array<GeneratedImage> => {
|
|
112
|
+
const revisedPrompt = item.revised_prompt
|
|
113
|
+
if (item.b64_json) {
|
|
114
|
+
return [{ b64Json: item.b64_json, revisedPrompt }]
|
|
115
|
+
}
|
|
116
|
+
if (item.url) {
|
|
117
|
+
return [{ url: item.url, revisedPrompt }]
|
|
118
|
+
}
|
|
119
|
+
return []
|
|
120
|
+
},
|
|
121
|
+
)
|
|
113
122
|
|
|
114
123
|
return {
|
|
115
124
|
id: generateId(this.name),
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'
|
|
2
|
+
import { generateId, getGrokApiKeyFromEnv, toAudioFile } from '../utils'
|
|
3
|
+
import type {
|
|
4
|
+
TranscriptionOptions,
|
|
5
|
+
TranscriptionResult,
|
|
6
|
+
TranscriptionWord,
|
|
7
|
+
} from '@tanstack/ai'
|
|
8
|
+
import type { GrokTranscriptionModel } from '../model-meta'
|
|
9
|
+
import type { GrokTranscriptionProviderOptions } from '../audio/transcription-provider-options'
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Grok-specific extension of `TranscriptionWord` that surfaces the extra
|
|
13
|
+
* fields xAI returns when diarization / confidence are enabled. The base
|
|
14
|
+
* cross-provider `TranscriptionWord` contract doesn't include these, so
|
|
15
|
+
* callers who know they're using Grok can narrow with:
|
|
16
|
+
*
|
|
17
|
+
* ```ts
|
|
18
|
+
* const words = result.words as Array<GrokTranscriptionWord> | undefined
|
|
19
|
+
* ```
|
|
20
|
+
*/
|
|
21
|
+
export interface GrokTranscriptionWord extends TranscriptionWord {
|
|
22
|
+
/** Model confidence for the word, when xAI returns one. */
|
|
23
|
+
confidence?: number
|
|
24
|
+
/** Speaker index, populated when `modelOptions.diarize === true`. */
|
|
25
|
+
speaker?: number
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
const DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Configuration for the Grok transcription adapter.
|
|
32
|
+
*
|
|
33
|
+
* Uses direct `fetch` rather than the OpenAI SDK because xAI's `/v1/stt`
|
|
34
|
+
* endpoint is not OpenAI-compatible.
|
|
35
|
+
*/
|
|
36
|
+
export interface GrokTranscriptionConfig {
|
|
37
|
+
apiKey: string
|
|
38
|
+
baseURL?: string
|
|
39
|
+
/** Additional headers to merge into every request (e.g., test IDs). */
|
|
40
|
+
defaultHeaders?: Record<string, string>
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* xAI STT response shape from `POST /v1/stt`.
|
|
45
|
+
* Grok returns word-level timestamps only; no segment array.
|
|
46
|
+
*/
|
|
47
|
+
interface GrokSTTWord {
|
|
48
|
+
text: string
|
|
49
|
+
start: number
|
|
50
|
+
end: number
|
|
51
|
+
confidence?: number
|
|
52
|
+
speaker?: number
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
interface GrokSTTResponse {
|
|
56
|
+
text: string
|
|
57
|
+
language?: string
|
|
58
|
+
duration?: number
|
|
59
|
+
words?: Array<GrokSTTWord>
|
|
60
|
+
channels?: Array<unknown>
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Grok Speech-to-Text Adapter.
|
|
65
|
+
*
|
|
66
|
+
* Talks to `POST {baseURL}/stt` per
|
|
67
|
+
* https://docs.x.ai/developers/rest-api-reference/inference/voice
|
|
68
|
+
*/
|
|
69
|
+
export class GrokTranscriptionAdapter<
|
|
70
|
+
TModel extends GrokTranscriptionModel,
|
|
71
|
+
> extends BaseTranscriptionAdapter<TModel, GrokTranscriptionProviderOptions> {
|
|
72
|
+
readonly name = 'grok' as const
|
|
73
|
+
|
|
74
|
+
private readonly apiKey: string
|
|
75
|
+
private readonly baseURL: string
|
|
76
|
+
private readonly defaultHeaders: Record<string, string>
|
|
77
|
+
|
|
78
|
+
constructor(config: GrokTranscriptionConfig, model: TModel) {
|
|
79
|
+
super(model, config)
|
|
80
|
+
this.apiKey = config.apiKey
|
|
81
|
+
this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, '')
|
|
82
|
+
this.defaultHeaders = config.defaultHeaders ?? {}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
async transcribe(
|
|
86
|
+
options: TranscriptionOptions<GrokTranscriptionProviderOptions>,
|
|
87
|
+
): Promise<TranscriptionResult> {
|
|
88
|
+
const { logger } = options
|
|
89
|
+
const { model, audio, language, modelOptions } = options
|
|
90
|
+
|
|
91
|
+
logger.request(
|
|
92
|
+
`activity=generateTranscription provider=grok model=${model}`,
|
|
93
|
+
{ provider: 'grok', model },
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
const file = toAudioFile(audio, modelOptions?.audio_format)
|
|
97
|
+
const form = buildTranscriptionFormData({ file, language, modelOptions })
|
|
98
|
+
|
|
99
|
+
try {
|
|
100
|
+
const response = await fetch(`${this.baseURL}/stt`, {
|
|
101
|
+
method: 'POST',
|
|
102
|
+
headers: {
|
|
103
|
+
// `defaultHeaders` first so Authorization always wins.
|
|
104
|
+
...this.defaultHeaders,
|
|
105
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
106
|
+
},
|
|
107
|
+
body: form,
|
|
108
|
+
})
|
|
109
|
+
|
|
110
|
+
if (!response.ok) {
|
|
111
|
+
const errorText = await response.text()
|
|
112
|
+
throw new Error(
|
|
113
|
+
`Grok transcription request failed: ${response.status} ${errorText}`,
|
|
114
|
+
)
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
const data = (await response.json()) as GrokSTTResponse
|
|
118
|
+
|
|
119
|
+
const words: Array<TranscriptionWord> | undefined = data.words?.map(
|
|
120
|
+
(w) => {
|
|
121
|
+
// Construct a GrokTranscriptionWord so that `confidence` and
|
|
122
|
+
// `speaker` (when xAI returns them under `diarize` / confidence
|
|
123
|
+
// mode) are preserved on the result. The returned array is typed
|
|
124
|
+
// as `Array<TranscriptionWord>` per the cross-provider contract;
|
|
125
|
+
// callers who want the extras narrow via `as Array<GrokTranscriptionWord>`.
|
|
126
|
+
const tw: GrokTranscriptionWord = {
|
|
127
|
+
word: w.text,
|
|
128
|
+
start: w.start,
|
|
129
|
+
end: w.end,
|
|
130
|
+
}
|
|
131
|
+
if (w.confidence !== undefined) tw.confidence = w.confidence
|
|
132
|
+
if (w.speaker !== undefined) tw.speaker = w.speaker
|
|
133
|
+
return tw
|
|
134
|
+
},
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
return {
|
|
138
|
+
id: generateId(this.name),
|
|
139
|
+
model,
|
|
140
|
+
text: data.text,
|
|
141
|
+
language: data.language ?? language,
|
|
142
|
+
duration: data.duration,
|
|
143
|
+
words,
|
|
144
|
+
}
|
|
145
|
+
} catch (error) {
|
|
146
|
+
logger.errors('grok.transcribe fatal', {
|
|
147
|
+
error,
|
|
148
|
+
source: 'grok.transcribe',
|
|
149
|
+
})
|
|
150
|
+
throw error
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* Build the multipart/form-data body for `POST /v1/stt`, coercing SDK-level
|
|
157
|
+
* model options into xAI's wire format (booleans as `'true'`/`'false'`
|
|
158
|
+
* strings, numeric fields stringified, etc.).
|
|
159
|
+
*
|
|
160
|
+
* Wire-field mapping:
|
|
161
|
+
* - `modelOptions.inverse_text_normalization` → `format` (xAI's chosen
|
|
162
|
+
* wire-field name for the ITN boolean; the SDK surfaces it under the
|
|
163
|
+
* clearer `inverse_text_normalization` key).
|
|
164
|
+
* - `modelOptions.audio_format`, `sample_rate`, `multichannel`, `channels`,
|
|
165
|
+
* `diarize` map to same-named form fields.
|
|
166
|
+
*/
|
|
167
|
+
export function buildTranscriptionFormData(options: {
|
|
168
|
+
file: File
|
|
169
|
+
language: string | undefined
|
|
170
|
+
modelOptions: GrokTranscriptionProviderOptions | undefined
|
|
171
|
+
}): FormData {
|
|
172
|
+
const { file, language, modelOptions } = options
|
|
173
|
+
const form = new FormData()
|
|
174
|
+
form.set('file', file)
|
|
175
|
+
if (language) form.set('language', language)
|
|
176
|
+
if (modelOptions?.audio_format !== undefined) {
|
|
177
|
+
form.set('audio_format', modelOptions.audio_format)
|
|
178
|
+
}
|
|
179
|
+
if (modelOptions?.sample_rate !== undefined) {
|
|
180
|
+
form.set('sample_rate', String(modelOptions.sample_rate))
|
|
181
|
+
}
|
|
182
|
+
if (modelOptions?.inverse_text_normalization !== undefined) {
|
|
183
|
+
form.set(
|
|
184
|
+
'format',
|
|
185
|
+
modelOptions.inverse_text_normalization ? 'true' : 'false',
|
|
186
|
+
)
|
|
187
|
+
}
|
|
188
|
+
if (modelOptions?.multichannel !== undefined) {
|
|
189
|
+
form.set('multichannel', modelOptions.multichannel ? 'true' : 'false')
|
|
190
|
+
}
|
|
191
|
+
if (modelOptions?.channels !== undefined) {
|
|
192
|
+
form.set('channels', String(modelOptions.channels))
|
|
193
|
+
}
|
|
194
|
+
if (modelOptions?.diarize !== undefined) {
|
|
195
|
+
form.set('diarize', modelOptions.diarize ? 'true' : 'false')
|
|
196
|
+
}
|
|
197
|
+
return form
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/**
|
|
201
|
+
* Creates a Grok transcription adapter with an explicit API key.
|
|
202
|
+
*
|
|
203
|
+
* @example
|
|
204
|
+
* ```typescript
|
|
205
|
+
* const adapter = createGrokTranscription('grok-stt', 'xai-...')
|
|
206
|
+
* const result = await generateTranscription({
|
|
207
|
+
* adapter,
|
|
208
|
+
* audio: audioFile,
|
|
209
|
+
* language: 'en',
|
|
210
|
+
* })
|
|
211
|
+
* ```
|
|
212
|
+
*/
|
|
213
|
+
export function createGrokTranscription<TModel extends GrokTranscriptionModel>(
|
|
214
|
+
model: TModel,
|
|
215
|
+
apiKey: string,
|
|
216
|
+
config?: Omit<GrokTranscriptionConfig, 'apiKey'>,
|
|
217
|
+
): GrokTranscriptionAdapter<TModel> {
|
|
218
|
+
return new GrokTranscriptionAdapter({ apiKey, ...config }, model)
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Creates a Grok transcription adapter, reading the API key from
|
|
223
|
+
* `XAI_API_KEY` in the environment.
|
|
224
|
+
*
|
|
225
|
+
* @throws Error if `XAI_API_KEY` is not set.
|
|
226
|
+
*/
|
|
227
|
+
export function grokTranscription<TModel extends GrokTranscriptionModel>(
|
|
228
|
+
model: TModel,
|
|
229
|
+
config?: Omit<GrokTranscriptionConfig, 'apiKey'>,
|
|
230
|
+
): GrokTranscriptionAdapter<TModel> {
|
|
231
|
+
const apiKey = getGrokApiKeyFromEnv()
|
|
232
|
+
return createGrokTranscription(model, apiKey, config)
|
|
233
|
+
}
|
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
import { BaseTTSAdapter } from '@tanstack/ai/adapters'
|
|
2
|
+
import { arrayBufferToBase64, generateId, getGrokApiKeyFromEnv } from '../utils'
|
|
3
|
+
import type { TTSOptions, TTSResult } from '@tanstack/ai'
|
|
4
|
+
import type { GrokTTSModel } from '../model-meta'
|
|
5
|
+
import type {
|
|
6
|
+
GrokTTSCodec,
|
|
7
|
+
GrokTTSProviderOptions,
|
|
8
|
+
GrokTTSVoice,
|
|
9
|
+
} from '../audio/tts-provider-options'
|
|
10
|
+
|
|
11
|
+
const DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Configuration for the Grok TTS adapter.
|
|
15
|
+
*
|
|
16
|
+
* Unlike chat/image/summarize adapters, TTS does not use the OpenAI SDK
|
|
17
|
+
* because xAI's `/v1/tts` endpoint is not OpenAI-compatible. This config
|
|
18
|
+
* is a minimal subset suitable for direct `fetch` calls.
|
|
19
|
+
*/
|
|
20
|
+
export interface GrokSpeechConfig {
|
|
21
|
+
apiKey: string
|
|
22
|
+
baseURL?: string
|
|
23
|
+
/** Additional headers to merge into every request (e.g., test IDs). */
|
|
24
|
+
defaultHeaders?: Record<string, string>
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Grok Text-to-Speech Adapter.
|
|
29
|
+
*
|
|
30
|
+
* Talks to `POST {baseURL}/tts` per
|
|
31
|
+
* https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
|
|
32
|
+
*/
|
|
33
|
+
export class GrokSpeechAdapter<
|
|
34
|
+
TModel extends GrokTTSModel,
|
|
35
|
+
> extends BaseTTSAdapter<TModel, GrokTTSProviderOptions> {
|
|
36
|
+
readonly name = 'grok' as const
|
|
37
|
+
|
|
38
|
+
private readonly apiKey: string
|
|
39
|
+
private readonly baseURL: string
|
|
40
|
+
private readonly defaultHeaders: Record<string, string>
|
|
41
|
+
|
|
42
|
+
constructor(config: GrokSpeechConfig, model: TModel) {
|
|
43
|
+
super(model, config)
|
|
44
|
+
this.apiKey = config.apiKey
|
|
45
|
+
this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, '')
|
|
46
|
+
this.defaultHeaders = config.defaultHeaders ?? {}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
async generateSpeech(
|
|
50
|
+
options: TTSOptions<GrokTTSProviderOptions>,
|
|
51
|
+
): Promise<TTSResult> {
|
|
52
|
+
const { logger } = options
|
|
53
|
+
const { model, text, voice, format, modelOptions } = options
|
|
54
|
+
|
|
55
|
+
logger.request(`activity=generateSpeech provider=grok model=${model}`, {
|
|
56
|
+
provider: 'grok',
|
|
57
|
+
model,
|
|
58
|
+
})
|
|
59
|
+
|
|
60
|
+
const { body, codec, sampleRateForContentType } = buildTTSRequestBody({
|
|
61
|
+
text,
|
|
62
|
+
voice,
|
|
63
|
+
format,
|
|
64
|
+
modelOptions,
|
|
65
|
+
})
|
|
66
|
+
|
|
67
|
+
try {
|
|
68
|
+
const response = await fetch(`${this.baseURL}/tts`, {
|
|
69
|
+
method: 'POST',
|
|
70
|
+
headers: {
|
|
71
|
+
// `defaultHeaders` first so the adapter's Authorization / Content-Type
|
|
72
|
+
// always win — otherwise a caller-supplied `Authorization` header
|
|
73
|
+
// could silently clobber the bearer token.
|
|
74
|
+
...this.defaultHeaders,
|
|
75
|
+
Authorization: `Bearer ${this.apiKey}`,
|
|
76
|
+
'Content-Type': 'application/json',
|
|
77
|
+
},
|
|
78
|
+
body: JSON.stringify(body),
|
|
79
|
+
})
|
|
80
|
+
|
|
81
|
+
if (!response.ok) {
|
|
82
|
+
const errorText = await response.text()
|
|
83
|
+
throw new Error(
|
|
84
|
+
`Grok TTS request failed: ${response.status} ${errorText}`,
|
|
85
|
+
)
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
const arrayBuffer = await response.arrayBuffer()
|
|
89
|
+
const audio = arrayBufferToBase64(arrayBuffer)
|
|
90
|
+
|
|
91
|
+
return {
|
|
92
|
+
id: generateId(this.name),
|
|
93
|
+
model,
|
|
94
|
+
audio,
|
|
95
|
+
format: codec,
|
|
96
|
+
contentType: getContentType(codec, sampleRateForContentType),
|
|
97
|
+
}
|
|
98
|
+
} catch (error) {
|
|
99
|
+
logger.errors('grok.generateSpeech fatal', {
|
|
100
|
+
error,
|
|
101
|
+
source: 'grok.generateSpeech',
|
|
102
|
+
})
|
|
103
|
+
throw error
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Build the JSON body for `POST /v1/tts`, resolving codec / sample-rate / voice
|
|
110
|
+
* defaults in one place.
|
|
111
|
+
*
|
|
112
|
+
* Returns the request `body`, the resolved `codec`, and the `sampleRateForContentType`
|
|
113
|
+
* used by the caller to label the response via `getContentType`.
|
|
114
|
+
*/
|
|
115
|
+
export function buildTTSRequestBody(options: {
|
|
116
|
+
text: string
|
|
117
|
+
voice: string | undefined
|
|
118
|
+
format: TTSOptions['format'] | undefined
|
|
119
|
+
modelOptions: GrokTTSProviderOptions | undefined
|
|
120
|
+
}): {
|
|
121
|
+
body: Record<string, unknown>
|
|
122
|
+
codec: GrokTTSCodec
|
|
123
|
+
sampleRateForContentType: number
|
|
124
|
+
} {
|
|
125
|
+
const { text, voice, format, modelOptions } = options
|
|
126
|
+
|
|
127
|
+
const codec = pickCodec(modelOptions?.codec, format)
|
|
128
|
+
|
|
129
|
+
// Only forward `sample_rate` when either:
|
|
130
|
+
// - the caller explicitly set `modelOptions.sample_rate`, or
|
|
131
|
+
// - the codec's Content-Type carries the rate (pcm → audio/L16;rate=…).
|
|
132
|
+
// For mp3/wav/opus/aac/flac we leave sample_rate unset so xAI's server
|
|
133
|
+
// default applies.
|
|
134
|
+
const callerSampleRate = modelOptions?.sample_rate
|
|
135
|
+
// Default sample rate documented in GrokTTSProviderOptions is 24000 Hz —
|
|
136
|
+
// used only when we MUST attach a rate to the contentType (pcm) and the
|
|
137
|
+
// caller didn't pick one.
|
|
138
|
+
const pcmDefault = 24000
|
|
139
|
+
const needsRateInContentType = codec === 'pcm'
|
|
140
|
+
|
|
141
|
+
const outputFormat: Record<string, unknown> = { codec }
|
|
142
|
+
if (callerSampleRate !== undefined) {
|
|
143
|
+
outputFormat.sample_rate = callerSampleRate
|
|
144
|
+
} else if (needsRateInContentType) {
|
|
145
|
+
outputFormat.sample_rate = pcmDefault
|
|
146
|
+
}
|
|
147
|
+
if (codec === 'mp3' && modelOptions?.bit_rate !== undefined) {
|
|
148
|
+
outputFormat.bit_rate = modelOptions.bit_rate
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
// pcm embeds the rate in `audio/L16;rate=…`; mulaw/alaw embed it in
|
|
152
|
+
// `audio/PCMU;rate=…` / `audio/PCMA;rate=…` when non-default. mp3/wav
|
|
153
|
+
// don't carry a rate parameter so the value is unused for those.
|
|
154
|
+
const sampleRateForContentType = callerSampleRate ?? pcmDefault
|
|
155
|
+
|
|
156
|
+
const body: Record<string, unknown> = {
|
|
157
|
+
text,
|
|
158
|
+
voice_id: (voice as GrokTTSVoice | undefined) ?? 'eve',
|
|
159
|
+
language: modelOptions?.language ?? 'en',
|
|
160
|
+
output_format: outputFormat,
|
|
161
|
+
}
|
|
162
|
+
if (modelOptions?.optimize_streaming_latency !== undefined) {
|
|
163
|
+
body.optimize_streaming_latency = modelOptions.optimize_streaming_latency
|
|
164
|
+
}
|
|
165
|
+
if (modelOptions?.text_normalization !== undefined) {
|
|
166
|
+
body.text_normalization = modelOptions.text_normalization
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
return { body, codec, sampleRateForContentType }
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Maps the cross-provider `TTSOptions.format` onto Grok's supported codecs.
|
|
174
|
+
* `opus`, `aac`, and `flac` are not supported by xAI TTS (which only exposes
|
|
175
|
+
* mp3/wav/pcm/mulaw/alaw) — we fall back to mp3. An explicit
|
|
176
|
+
* `modelOptions.codec` always wins.
|
|
177
|
+
*/
|
|
178
|
+
function pickCodec(
|
|
179
|
+
codecOverride: GrokTTSCodec | undefined,
|
|
180
|
+
format: TTSOptions['format'] | undefined,
|
|
181
|
+
): GrokTTSCodec {
|
|
182
|
+
if (codecOverride) return codecOverride
|
|
183
|
+
if (!format) return 'mp3'
|
|
184
|
+
switch (format) {
|
|
185
|
+
case 'mp3':
|
|
186
|
+
case 'wav':
|
|
187
|
+
case 'pcm':
|
|
188
|
+
return format
|
|
189
|
+
case 'flac':
|
|
190
|
+
case 'opus':
|
|
191
|
+
case 'aac':
|
|
192
|
+
return 'mp3'
|
|
193
|
+
default:
|
|
194
|
+
return 'mp3'
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
export function getContentType(
|
|
199
|
+
codec: GrokTTSCodec,
|
|
200
|
+
sampleRate: number,
|
|
201
|
+
): string {
|
|
202
|
+
switch (codec) {
|
|
203
|
+
case 'mp3':
|
|
204
|
+
return 'audio/mpeg'
|
|
205
|
+
case 'wav':
|
|
206
|
+
return 'audio/wav'
|
|
207
|
+
case 'pcm':
|
|
208
|
+
// `audio/L16` requires a `rate` parameter per RFC 3551/3555.
|
|
209
|
+
return `audio/L16;rate=${sampleRate}`
|
|
210
|
+
case 'mulaw':
|
|
211
|
+
// `audio/basic` is 8 kHz mono by RFC 2046 registration. For non-8kHz
|
|
212
|
+
// streams xAI still produces mulaw-encoded bytes at the requested
|
|
213
|
+
// rate, but the registered MIME can't carry that rate — so we use
|
|
214
|
+
// the non-standard but commonly-supported `audio/PCMU;rate=…` (RFC 3551
|
|
215
|
+
// RTP payload name) whenever the caller asked for a rate other than
|
|
216
|
+
// 8000, and keep `audio/basic` for the standard 8kHz case.
|
|
217
|
+
return sampleRate === 8000
|
|
218
|
+
? 'audio/basic'
|
|
219
|
+
: `audio/PCMU;rate=${sampleRate}`
|
|
220
|
+
case 'alaw':
|
|
221
|
+
return sampleRate === 8000
|
|
222
|
+
? 'audio/x-alaw-basic'
|
|
223
|
+
: `audio/PCMA;rate=${sampleRate}`
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* Creates a Grok speech (TTS) adapter with an explicit API key.
|
|
229
|
+
*
|
|
230
|
+
* @example
|
|
231
|
+
* ```typescript
|
|
232
|
+
* const adapter = createGrokSpeech('grok-tts', 'xai-...')
|
|
233
|
+
* const result = await generateSpeech({
|
|
234
|
+
* adapter,
|
|
235
|
+
* text: 'Hello from Grok',
|
|
236
|
+
* voice: 'eve',
|
|
237
|
+
* })
|
|
238
|
+
* ```
|
|
239
|
+
*/
|
|
240
|
+
export function createGrokSpeech<TModel extends GrokTTSModel>(
|
|
241
|
+
model: TModel,
|
|
242
|
+
apiKey: string,
|
|
243
|
+
config?: Omit<GrokSpeechConfig, 'apiKey'>,
|
|
244
|
+
): GrokSpeechAdapter<TModel> {
|
|
245
|
+
return new GrokSpeechAdapter({ apiKey, ...config }, model)
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Creates a Grok speech (TTS) adapter, reading the API key from
|
|
250
|
+
* `XAI_API_KEY` in the environment.
|
|
251
|
+
*
|
|
252
|
+
* @throws Error if `XAI_API_KEY` is not set.
|
|
253
|
+
*/
|
|
254
|
+
export function grokSpeech<TModel extends GrokTTSModel>(
|
|
255
|
+
model: TModel,
|
|
256
|
+
config?: Omit<GrokSpeechConfig, 'apiKey'>,
|
|
257
|
+
): GrokSpeechAdapter<TModel> {
|
|
258
|
+
const apiKey = getGrokApiKeyFromEnv()
|
|
259
|
+
return createGrokSpeech(model, apiKey, config)
|
|
260
|
+
}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Grok STT supported audio formats.
|
|
3
|
+
* See https://docs.x.ai/developers/rest-api-reference/inference/voice
|
|
4
|
+
*/
|
|
5
|
+
export type GrokSTTAudioFormat =
|
|
6
|
+
| 'pcm'
|
|
7
|
+
| 'mulaw'
|
|
8
|
+
| 'alaw'
|
|
9
|
+
| 'wav'
|
|
10
|
+
| 'mp3'
|
|
11
|
+
| 'ogg'
|
|
12
|
+
| 'opus'
|
|
13
|
+
| 'flac'
|
|
14
|
+
| 'aac'
|
|
15
|
+
| 'mp4'
|
|
16
|
+
| 'm4a'
|
|
17
|
+
| 'mkv'
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Provider-specific options for Grok transcription (`POST /v1/stt`).
|
|
21
|
+
*/
|
|
22
|
+
export interface GrokTranscriptionProviderOptions {
|
|
23
|
+
/**
|
|
24
|
+
* The format of the provided audio. Required for raw codecs (pcm, mulaw, alaw).
|
|
25
|
+
*/
|
|
26
|
+
audio_format?: GrokSTTAudioFormat
|
|
27
|
+
/**
|
|
28
|
+
* Sample rate of the audio (Hz). Required for raw codecs.
|
|
29
|
+
*/
|
|
30
|
+
sample_rate?: number
|
|
31
|
+
/**
|
|
32
|
+
* Apply inverse text normalization (e.g. "one hundred" → "100"). Requires
|
|
33
|
+
* `language` to be set on the core `TranscriptionOptions`.
|
|
34
|
+
*
|
|
35
|
+
* NOTE: xAI's STT API exposes this on the wire as `format` (a boolean
|
|
36
|
+
* toggle). We surface it under the clearer name
|
|
37
|
+
* `inverse_text_normalization` on the SDK, and translate to the wire name
|
|
38
|
+
* inside the adapter.
|
|
39
|
+
*/
|
|
40
|
+
inverse_text_normalization?: boolean
|
|
41
|
+
/**
|
|
42
|
+
* Treat the audio as multichannel. When enabled, `channels` must also be set.
|
|
43
|
+
*/
|
|
44
|
+
multichannel?: boolean
|
|
45
|
+
/**
|
|
46
|
+
* Channel count for multichannel raw audio (2–8).
|
|
47
|
+
*/
|
|
48
|
+
channels?: number
|
|
49
|
+
/**
|
|
50
|
+
* Enable speaker diarization. When true, response words include a `speaker`
|
|
51
|
+
* field.
|
|
52
|
+
*/
|
|
53
|
+
diarize?: boolean
|
|
54
|
+
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Grok TTS voice options.
|
|
3
|
+
* See https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
|
|
4
|
+
*/
|
|
5
|
+
export type GrokTTSVoice = 'eve' | 'ara' | 'rex' | 'sal' | 'leo'
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Grok TTS output audio codecs.
|
|
9
|
+
* Grok does NOT support opus or aac; those formats are mapped to mp3.
|
|
10
|
+
*/
|
|
11
|
+
export type GrokTTSCodec = 'mp3' | 'wav' | 'pcm' | 'mulaw' | 'alaw'
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Provider-specific options for Grok TTS (`POST /v1/tts`).
|
|
15
|
+
*/
|
|
16
|
+
export interface GrokTTSProviderOptions {
|
|
17
|
+
/**
|
|
18
|
+
* BCP-47 language code (e.g., `en`, `zh`, `pt-BR`) or `'auto'` for detection.
|
|
19
|
+
* Defaults to `'en'` when not provided.
|
|
20
|
+
*/
|
|
21
|
+
language?: string
|
|
22
|
+
/**
|
|
23
|
+
* Audio codec. Overrides the `format` field on `TTSOptions` when set.
|
|
24
|
+
*/
|
|
25
|
+
codec?: GrokTTSCodec
|
|
26
|
+
/**
|
|
27
|
+
* Sample rate in Hz. Valid values: 8000, 16000, 22050, 24000, 44100, 48000.
|
|
28
|
+
* Defaults to 24000.
|
|
29
|
+
*/
|
|
30
|
+
sample_rate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000
|
|
31
|
+
/**
|
|
32
|
+
* Bit rate for MP3 output. Ignored for other codecs.
|
|
33
|
+
* Valid values: 32000, 64000, 96000, 128000, 192000. Defaults to 128000.
|
|
34
|
+
*/
|
|
35
|
+
bit_rate?: 32000 | 64000 | 96000 | 128000 | 192000
|
|
36
|
+
/**
|
|
37
|
+
* Set to 1 for lower latency streaming; 0 (default) for normal quality.
|
|
38
|
+
*/
|
|
39
|
+
optimize_streaming_latency?: 0 | 1
|
|
40
|
+
/**
|
|
41
|
+
* Enable text normalization. Defaults to false.
|
|
42
|
+
*/
|
|
43
|
+
text_normalization?: boolean
|
|
44
|
+
}
|