@tanstack/ai-grok 0.6.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/dist/esm/adapters/image.js +36 -17
  2. package/dist/esm/adapters/image.js.map +1 -1
  3. package/dist/esm/adapters/summarize.js +51 -22
  4. package/dist/esm/adapters/summarize.js.map +1 -1
  5. package/dist/esm/adapters/text.js +25 -10
  6. package/dist/esm/adapters/text.js.map +1 -1
  7. package/dist/esm/adapters/transcription.d.ts +84 -0
  8. package/dist/esm/adapters/transcription.js +109 -0
  9. package/dist/esm/adapters/transcription.js.map +1 -0
  10. package/dist/esm/adapters/tts.d.ts +70 -0
  11. package/dist/esm/adapters/tts.js +137 -0
  12. package/dist/esm/adapters/tts.js.map +1 -0
  13. package/dist/esm/audio/transcription-provider-options.d.ts +41 -0
  14. package/dist/esm/audio/tts-provider-options.d.ts +42 -0
  15. package/dist/esm/index.d.ts +8 -2
  16. package/dist/esm/index.js +17 -2
  17. package/dist/esm/index.js.map +1 -1
  18. package/dist/esm/model-meta.d.ts +6 -0
  19. package/dist/esm/model-meta.js +22 -1
  20. package/dist/esm/model-meta.js.map +1 -1
  21. package/dist/esm/realtime/adapter.d.ts +21 -0
  22. package/dist/esm/realtime/adapter.js +816 -0
  23. package/dist/esm/realtime/adapter.js.map +1 -0
  24. package/dist/esm/realtime/index.d.ts +4 -0
  25. package/dist/esm/realtime/realtime-contract.d.ts +30 -0
  26. package/dist/esm/realtime/token.d.ts +22 -0
  27. package/dist/esm/realtime/token.js +73 -0
  28. package/dist/esm/realtime/token.js.map +1 -0
  29. package/dist/esm/realtime/types.d.ts +95 -0
  30. package/dist/esm/utils/audio.d.ts +23 -0
  31. package/dist/esm/utils/audio.js +171 -0
  32. package/dist/esm/utils/audio.js.map +1 -0
  33. package/dist/esm/utils/index.d.ts +1 -0
  34. package/package.json +6 -3
  35. package/src/adapters/image.ts +41 -19
  36. package/src/adapters/summarize.ts +56 -25
  37. package/src/adapters/text.ts +26 -9
  38. package/src/adapters/transcription.ts +233 -0
  39. package/src/adapters/tts.ts +260 -0
  40. package/src/audio/transcription-provider-options.ts +54 -0
  41. package/src/audio/tts-provider-options.ts +44 -0
  42. package/src/index.ts +50 -1
  43. package/src/model-meta.ts +54 -0
  44. package/src/realtime/adapter.ts +1215 -0
  45. package/src/realtime/index.ts +18 -0
  46. package/src/realtime/realtime-contract.ts +46 -0
  47. package/src/realtime/token.ts +131 -0
  48. package/src/realtime/types.ts +105 -0
  49. package/src/utils/audio.ts +217 -0
  50. package/src/utils/index.ts +1 -0
@@ -0,0 +1,109 @@
1
+ import { BaseTranscriptionAdapter } from "@tanstack/ai/adapters";
2
+ import { generateId, getGrokApiKeyFromEnv } from "../utils/client.js";
3
+ import { toAudioFile } from "../utils/audio.js";
4
+ const DEFAULT_GROK_BASE_URL = "https://api.x.ai/v1";
5
+ class GrokTranscriptionAdapter extends BaseTranscriptionAdapter {
6
+ constructor(config, model) {
7
+ super(model, config);
8
+ this.name = "grok";
9
+ this.apiKey = config.apiKey;
10
+ this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, "");
11
+ this.defaultHeaders = config.defaultHeaders ?? {};
12
+ }
13
+ async transcribe(options) {
14
+ const { logger } = options;
15
+ const { model, audio, language, modelOptions } = options;
16
+ logger.request(
17
+ `activity=generateTranscription provider=grok model=${model}`,
18
+ { provider: "grok", model }
19
+ );
20
+ const file = toAudioFile(audio, modelOptions?.audio_format);
21
+ const form = buildTranscriptionFormData({ file, language, modelOptions });
22
+ try {
23
+ const response = await fetch(`${this.baseURL}/stt`, {
24
+ method: "POST",
25
+ headers: {
26
+ // `defaultHeaders` first so Authorization always wins.
27
+ ...this.defaultHeaders,
28
+ Authorization: `Bearer ${this.apiKey}`
29
+ },
30
+ body: form
31
+ });
32
+ if (!response.ok) {
33
+ const errorText = await response.text();
34
+ throw new Error(
35
+ `Grok transcription request failed: ${response.status} ${errorText}`
36
+ );
37
+ }
38
+ const data = await response.json();
39
+ const words = data.words?.map(
40
+ (w) => {
41
+ const tw = {
42
+ word: w.text,
43
+ start: w.start,
44
+ end: w.end
45
+ };
46
+ if (w.confidence !== void 0) tw.confidence = w.confidence;
47
+ if (w.speaker !== void 0) tw.speaker = w.speaker;
48
+ return tw;
49
+ }
50
+ );
51
+ return {
52
+ id: generateId(this.name),
53
+ model,
54
+ text: data.text,
55
+ language: data.language ?? language,
56
+ duration: data.duration,
57
+ words
58
+ };
59
+ } catch (error) {
60
+ logger.errors("grok.transcribe fatal", {
61
+ error,
62
+ source: "grok.transcribe"
63
+ });
64
+ throw error;
65
+ }
66
+ }
67
+ }
68
+ function buildTranscriptionFormData(options) {
69
+ const { file, language, modelOptions } = options;
70
+ const form = new FormData();
71
+ form.set("file", file);
72
+ if (language) form.set("language", language);
73
+ if (modelOptions?.audio_format !== void 0) {
74
+ form.set("audio_format", modelOptions.audio_format);
75
+ }
76
+ if (modelOptions?.sample_rate !== void 0) {
77
+ form.set("sample_rate", String(modelOptions.sample_rate));
78
+ }
79
+ if (modelOptions?.inverse_text_normalization !== void 0) {
80
+ form.set(
81
+ "format",
82
+ modelOptions.inverse_text_normalization ? "true" : "false"
83
+ );
84
+ }
85
+ if (modelOptions?.multichannel !== void 0) {
86
+ form.set("multichannel", modelOptions.multichannel ? "true" : "false");
87
+ }
88
+ if (modelOptions?.channels !== void 0) {
89
+ form.set("channels", String(modelOptions.channels));
90
+ }
91
+ if (modelOptions?.diarize !== void 0) {
92
+ form.set("diarize", modelOptions.diarize ? "true" : "false");
93
+ }
94
+ return form;
95
+ }
96
+ function createGrokTranscription(model, apiKey, config) {
97
+ return new GrokTranscriptionAdapter({ apiKey, ...config }, model);
98
+ }
99
+ function grokTranscription(model, config) {
100
+ const apiKey = getGrokApiKeyFromEnv();
101
+ return createGrokTranscription(model, apiKey, config);
102
+ }
103
+ export {
104
+ GrokTranscriptionAdapter,
105
+ buildTranscriptionFormData,
106
+ createGrokTranscription,
107
+ grokTranscription
108
+ };
109
+ //# sourceMappingURL=transcription.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"transcription.js","sources":["../../../src/adapters/transcription.ts"],"sourcesContent":["import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'\nimport { generateId, getGrokApiKeyFromEnv, toAudioFile } from '../utils'\nimport type {\n TranscriptionOptions,\n TranscriptionResult,\n TranscriptionWord,\n} from '@tanstack/ai'\nimport type { GrokTranscriptionModel } from '../model-meta'\nimport type { GrokTranscriptionProviderOptions } from '../audio/transcription-provider-options'\n\n/**\n * Grok-specific extension of `TranscriptionWord` that surfaces the extra\n * fields xAI returns when diarization / confidence are enabled. The base\n * cross-provider `TranscriptionWord` contract doesn't include these, so\n * callers who know they're using Grok can narrow with:\n *\n * ```ts\n * const words = result.words as Array<GrokTranscriptionWord> | undefined\n * ```\n */\nexport interface GrokTranscriptionWord extends TranscriptionWord {\n /** Model confidence for the word, when xAI returns one. */\n confidence?: number\n /** Speaker index, populated when `modelOptions.diarize === true`. */\n speaker?: number\n}\n\nconst DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'\n\n/**\n * Configuration for the Grok transcription adapter.\n *\n * Uses direct `fetch` rather than the OpenAI SDK because xAI's `/v1/stt`\n * endpoint is not OpenAI-compatible.\n */\nexport interface GrokTranscriptionConfig {\n apiKey: string\n baseURL?: string\n /** Additional headers to merge into every request (e.g., test IDs). */\n defaultHeaders?: Record<string, string>\n}\n\n/**\n * xAI STT response shape from `POST /v1/stt`.\n * Grok returns word-level timestamps only; no segment array.\n */\ninterface GrokSTTWord {\n text: string\n start: number\n end: number\n confidence?: number\n speaker?: number\n}\n\ninterface GrokSTTResponse {\n text: string\n language?: string\n duration?: number\n words?: Array<GrokSTTWord>\n channels?: Array<unknown>\n}\n\n/**\n * Grok Speech-to-Text Adapter.\n *\n * Talks to `POST {baseURL}/stt` per\n * https://docs.x.ai/developers/rest-api-reference/inference/voice\n */\nexport class GrokTranscriptionAdapter<\n TModel extends GrokTranscriptionModel,\n> extends BaseTranscriptionAdapter<TModel, GrokTranscriptionProviderOptions> {\n readonly name = 'grok' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n\n constructor(config: GrokTranscriptionConfig, model: TModel) {\n super(model, config)\n this.apiKey = config.apiKey\n this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\\/+$/, '')\n this.defaultHeaders = config.defaultHeaders ?? {}\n }\n\n async transcribe(\n options: TranscriptionOptions<GrokTranscriptionProviderOptions>,\n ): Promise<TranscriptionResult> {\n const { logger } = options\n const { model, audio, language, modelOptions } = options\n\n logger.request(\n `activity=generateTranscription provider=grok model=${model}`,\n { provider: 'grok', model },\n )\n\n const file = toAudioFile(audio, modelOptions?.audio_format)\n const form = buildTranscriptionFormData({ file, language, modelOptions })\n\n try {\n const response = await fetch(`${this.baseURL}/stt`, {\n method: 'POST',\n headers: {\n // `defaultHeaders` first so Authorization always wins.\n ...this.defaultHeaders,\n Authorization: `Bearer ${this.apiKey}`,\n },\n body: form,\n })\n\n if (!response.ok) {\n const errorText = await response.text()\n throw new Error(\n `Grok transcription request failed: ${response.status} ${errorText}`,\n )\n }\n\n const data = (await response.json()) as GrokSTTResponse\n\n const words: Array<TranscriptionWord> | undefined = data.words?.map(\n (w) => {\n // Construct a GrokTranscriptionWord so that `confidence` and\n // `speaker` (when xAI returns them under `diarize` / confidence\n // mode) are preserved on the result. The returned array is typed\n // as `Array<TranscriptionWord>` per the cross-provider contract;\n // callers who want the extras narrow via `as Array<GrokTranscriptionWord>`.\n const tw: GrokTranscriptionWord = {\n word: w.text,\n start: w.start,\n end: w.end,\n }\n if (w.confidence !== undefined) tw.confidence = w.confidence\n if (w.speaker !== undefined) tw.speaker = w.speaker\n return tw\n },\n )\n\n return {\n id: generateId(this.name),\n model,\n text: data.text,\n language: data.language ?? language,\n duration: data.duration,\n words,\n }\n } catch (error) {\n logger.errors('grok.transcribe fatal', {\n error,\n source: 'grok.transcribe',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the multipart/form-data body for `POST /v1/stt`, coercing SDK-level\n * model options into xAI's wire format (booleans as `'true'`/`'false'`\n * strings, numeric fields stringified, etc.).\n *\n * Wire-field mapping:\n * - `modelOptions.inverse_text_normalization` → `format` (xAI's chosen\n * wire-field name for the ITN boolean; the SDK surfaces it under the\n * clearer `inverse_text_normalization` key).\n * - `modelOptions.audio_format`, `sample_rate`, `multichannel`, `channels`,\n * `diarize` map to same-named form fields.\n */\nexport function buildTranscriptionFormData(options: {\n file: File\n language: string | undefined\n modelOptions: GrokTranscriptionProviderOptions | undefined\n}): FormData {\n const { file, language, modelOptions } = options\n const form = new FormData()\n form.set('file', file)\n if (language) form.set('language', language)\n if (modelOptions?.audio_format !== undefined) {\n form.set('audio_format', modelOptions.audio_format)\n }\n if (modelOptions?.sample_rate !== undefined) {\n form.set('sample_rate', String(modelOptions.sample_rate))\n }\n if (modelOptions?.inverse_text_normalization !== undefined) {\n form.set(\n 'format',\n modelOptions.inverse_text_normalization ? 'true' : 'false',\n )\n }\n if (modelOptions?.multichannel !== undefined) {\n form.set('multichannel', modelOptions.multichannel ? 'true' : 'false')\n }\n if (modelOptions?.channels !== undefined) {\n form.set('channels', String(modelOptions.channels))\n }\n if (modelOptions?.diarize !== undefined) {\n form.set('diarize', modelOptions.diarize ? 'true' : 'false')\n }\n return form\n}\n\n/**\n * Creates a Grok transcription adapter with an explicit API key.\n *\n * @example\n * ```typescript\n * const adapter = createGrokTranscription('grok-stt', 'xai-...')\n * const result = await generateTranscription({\n * adapter,\n * audio: audioFile,\n * language: 'en',\n * })\n * ```\n */\nexport function createGrokTranscription<TModel extends GrokTranscriptionModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GrokTranscriptionConfig, 'apiKey'>,\n): GrokTranscriptionAdapter<TModel> {\n return new GrokTranscriptionAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Grok transcription adapter, reading the API key from\n * `XAI_API_KEY` in the environment.\n *\n * @throws Error if `XAI_API_KEY` is not set.\n */\nexport function grokTranscription<TModel extends GrokTranscriptionModel>(\n model: TModel,\n config?: Omit<GrokTranscriptionConfig, 'apiKey'>,\n): GrokTranscriptionAdapter<TModel> {\n const apiKey = getGrokApiKeyFromEnv()\n return createGrokTranscription(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;AA2BA,MAAM,wBAAwB;AAyCvB,MAAM,iCAEH,yBAAmE;AAAA,EAO3E,YAAY,QAAiC,OAAe;AAC1D,UAAM,OAAO,MAAM;AAPrB,SAAS,OAAO;AAQd,SAAK,SAAS,OAAO;AACrB,SAAK,WAAW,OAAO,WAAW,uBAAuB,QAAQ,QAAQ,EAAE;AAC3E,SAAK,iBAAiB,OAAO,kBAAkB,CAAA;AAAA,EACjD;AAAA,EAEA,MAAM,WACJ,SAC8B;AAC9B,UAAM,EAAE,WAAW;AACnB,UAAM,EAAE,OAAO,OAAO,UAAU,iBAAiB;AAEjD,WAAO;AAAA,MACL,sDAAsD,KAAK;AAAA,MAC3D,EAAE,UAAU,QAAQ,MAAA;AAAA,IAAM;AAG5B,UAAM,OAAO,YAAY,OAAO,cAAc,YAAY;AAC1D,UAAM,OAAO,2BAA2B,EAAE,MAAM,UAAU,cAAc;AAExE,QAAI;AACF,YAAM,WAAW,MAAM,MAAM,GAAG,KAAK,OAAO,QAAQ;AAAA,QAClD,QAAQ;AAAA,QACR,SAAS;AAAA;AAAA,UAEP,GAAG,KAAK;AAAA,UACR,eAAe,UAAU,KAAK,MAAM;AAAA,QAAA;AAAA,QAEtC,MAAM;AAAA,MAAA,CACP;AAED,UAAI,CAAC,SAAS,IAAI;AAChB,cAAM,YAAY,MAAM,SAAS,KAAA;AACjC,cAAM,IAAI;AAAA,UACR,sCAAsC,SAAS,MAAM,IAAI,SAAS;AAAA,QAAA;AAAA,MAEtE;AAEA,YAAM,OAAQ,MAAM,SAAS,KAAA;AAE7B,YAAM,QAA8C,KAAK,OAAO;AAAA,QAC9D,CAAC,MAAM;AAML,gBAAM,KAA4B;AAAA,YAChC,MAAM,EAAE;AAAA,YACR,OAAO,EAAE;AAAA,YACT,KAAK,EAAE;AAAA,UAAA;AAET,cAAI,EAAE,eAAe,OAAW,IAAG,aAAa,EAAE;AAClD,cAAI,EAAE,YAAY,OAAW,IAAG,UAAU,EAAE;AAC5C,iBAAO;AAAA,QACT;AAAA,MAAA;AAGF,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA,MAAM,KAAK;AAAA,QACX,UAAU,KAAK,YAAY;AAAA,QAC3B,UAAU,KAAK;AAAA,QACf;AAAA,MAAA;AAAA,IAEJ,SAAS,OAAO;AACd,aAAO,OAAO,yBAAyB;AAAA,QACrC;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AAcO,SAAS,2BAA2B,SAI9B;AACX,QAAM,EAAE,MAAM,UAAU,aAAA,IAAiB;AACzC,QAAM,OAAO,IAAI,SAAA;AACjB,OAAK,IAAI,QAAQ,IAAI;AACrB,MAAI,SAAU,MAAK,IAAI,YAAY,QAAQ;AAC3C,MAAI,cAAc,iBAAiB,QAAW;AAC5C,SAAK,IAAI,gBAAgB,aAAa,YAAY;AAAA,EACpD;AACA,MAAI,cAAc,gBAAgB,QAAW;AAC3C,SAAK,IAAI,eAAe,OAAO,aAAa,WAAW,CAAC;AAAA,EAC1D;AACA,MAAI,cAAc,+BAA+B,QAAW;AAC1D,SAAK;AAAA,MACH;AAAA,MACA,aAAa,6BAA6B,SAAS;AAAA,IAAA;AAAA,EAEvD;AACA,MAAI,cAAc,iBAAiB,QAAW;AAC5C,SAAK,IAAI,gBAAgB,aAAa,eAAe,SAAS,OAAO;AAAA,EACvE;AACA,MAAI,cAAc,aAAa,QAAW;AACxC,SAAK,IAAI,YAAY,OAAO,aAAa,QAAQ,CAAC;AAAA,EACpD;AACA,MAAI,cAAc,YAAY,QAAW;AACvC,SAAK,IAAI,WAAW,aAAa,UAAU,SAAS,OAAO;AAAA,EAC7D;AACA,SAAO;AACT;AAeO,SAAS,wBACd,OACA,QACA,QACkC;AAClC,SAAO,IAAI,yBAAyB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAClE;AAQO,SAAS,kBACd,OACA,QACkC;AAClC,QAAM,SAAS,qBAAA;AACf,SAAO,wBAAwB,OAAO,QAAQ,MAAM;AACtD;"}
@@ -0,0 +1,70 @@
1
+ import { BaseTTSAdapter } from '@tanstack/ai/adapters';
2
+ import { TTSOptions, TTSResult } from '@tanstack/ai';
3
+ import { GrokTTSModel } from '../model-meta.js';
4
+ import { GrokTTSCodec, GrokTTSProviderOptions } from '../audio/tts-provider-options.js';
5
+ /**
6
+ * Configuration for the Grok TTS adapter.
7
+ *
8
+ * Unlike chat/image/summarize adapters, TTS does not use the OpenAI SDK
9
+ * because xAI's `/v1/tts` endpoint is not OpenAI-compatible. This config
10
+ * is a minimal subset suitable for direct `fetch` calls.
11
+ */
12
+ export interface GrokSpeechConfig {
13
+ apiKey: string;
14
+ baseURL?: string;
15
+ /** Additional headers to merge into every request (e.g., test IDs). */
16
+ defaultHeaders?: Record<string, string>;
17
+ }
18
+ /**
19
+ * Grok Text-to-Speech Adapter.
20
+ *
21
+ * Talks to `POST {baseURL}/tts` per
22
+ * https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
23
+ */
24
+ export declare class GrokSpeechAdapter<TModel extends GrokTTSModel> extends BaseTTSAdapter<TModel, GrokTTSProviderOptions> {
25
+ readonly name: "grok";
26
+ private readonly apiKey;
27
+ private readonly baseURL;
28
+ private readonly defaultHeaders;
29
+ constructor(config: GrokSpeechConfig, model: TModel);
30
+ generateSpeech(options: TTSOptions<GrokTTSProviderOptions>): Promise<TTSResult>;
31
+ }
32
+ /**
33
+ * Build the JSON body for `POST /v1/tts`, resolving codec / sample-rate / voice
34
+ * defaults in one place.
35
+ *
36
+ * Returns the request `body`, the resolved `codec`, and the `sampleRateForContentType`
37
+ * used by the caller to label the response via `getContentType`.
38
+ */
39
+ export declare function buildTTSRequestBody(options: {
40
+ text: string;
41
+ voice: string | undefined;
42
+ format: TTSOptions['format'] | undefined;
43
+ modelOptions: GrokTTSProviderOptions | undefined;
44
+ }): {
45
+ body: Record<string, unknown>;
46
+ codec: GrokTTSCodec;
47
+ sampleRateForContentType: number;
48
+ };
49
+ export declare function getContentType(codec: GrokTTSCodec, sampleRate: number): string;
50
+ /**
51
+ * Creates a Grok speech (TTS) adapter with an explicit API key.
52
+ *
53
+ * @example
54
+ * ```typescript
55
+ * const adapter = createGrokSpeech('grok-tts', 'xai-...')
56
+ * const result = await generateSpeech({
57
+ * adapter,
58
+ * text: 'Hello from Grok',
59
+ * voice: 'eve',
60
+ * })
61
+ * ```
62
+ */
63
+ export declare function createGrokSpeech<TModel extends GrokTTSModel>(model: TModel, apiKey: string, config?: Omit<GrokSpeechConfig, 'apiKey'>): GrokSpeechAdapter<TModel>;
64
+ /**
65
+ * Creates a Grok speech (TTS) adapter, reading the API key from
66
+ * `XAI_API_KEY` in the environment.
67
+ *
68
+ * @throws Error if `XAI_API_KEY` is not set.
69
+ */
70
+ export declare function grokSpeech<TModel extends GrokTTSModel>(model: TModel, config?: Omit<GrokSpeechConfig, 'apiKey'>): GrokSpeechAdapter<TModel>;
@@ -0,0 +1,137 @@
1
+ import { BaseTTSAdapter } from "@tanstack/ai/adapters";
2
+ import { generateId, getGrokApiKeyFromEnv } from "../utils/client.js";
3
+ import { arrayBufferToBase64 } from "../utils/audio.js";
4
+ const DEFAULT_GROK_BASE_URL = "https://api.x.ai/v1";
5
+ class GrokSpeechAdapter extends BaseTTSAdapter {
6
+ constructor(config, model) {
7
+ super(model, config);
8
+ this.name = "grok";
9
+ this.apiKey = config.apiKey;
10
+ this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, "");
11
+ this.defaultHeaders = config.defaultHeaders ?? {};
12
+ }
13
+ async generateSpeech(options) {
14
+ const { logger } = options;
15
+ const { model, text, voice, format, modelOptions } = options;
16
+ logger.request(`activity=generateSpeech provider=grok model=${model}`, {
17
+ provider: "grok",
18
+ model
19
+ });
20
+ const { body, codec, sampleRateForContentType } = buildTTSRequestBody({
21
+ text,
22
+ voice,
23
+ format,
24
+ modelOptions
25
+ });
26
+ try {
27
+ const response = await fetch(`${this.baseURL}/tts`, {
28
+ method: "POST",
29
+ headers: {
30
+ // `defaultHeaders` first so the adapter's Authorization / Content-Type
31
+ // always win — otherwise a caller-supplied `Authorization` header
32
+ // could silently clobber the bearer token.
33
+ ...this.defaultHeaders,
34
+ Authorization: `Bearer ${this.apiKey}`,
35
+ "Content-Type": "application/json"
36
+ },
37
+ body: JSON.stringify(body)
38
+ });
39
+ if (!response.ok) {
40
+ const errorText = await response.text();
41
+ throw new Error(
42
+ `Grok TTS request failed: ${response.status} ${errorText}`
43
+ );
44
+ }
45
+ const arrayBuffer = await response.arrayBuffer();
46
+ const audio = arrayBufferToBase64(arrayBuffer);
47
+ return {
48
+ id: generateId(this.name),
49
+ model,
50
+ audio,
51
+ format: codec,
52
+ contentType: getContentType(codec, sampleRateForContentType)
53
+ };
54
+ } catch (error) {
55
+ logger.errors("grok.generateSpeech fatal", {
56
+ error,
57
+ source: "grok.generateSpeech"
58
+ });
59
+ throw error;
60
+ }
61
+ }
62
+ }
63
+ function buildTTSRequestBody(options) {
64
+ const { text, voice, format, modelOptions } = options;
65
+ const codec = pickCodec(modelOptions?.codec, format);
66
+ const callerSampleRate = modelOptions?.sample_rate;
67
+ const pcmDefault = 24e3;
68
+ const needsRateInContentType = codec === "pcm";
69
+ const outputFormat = { codec };
70
+ if (callerSampleRate !== void 0) {
71
+ outputFormat.sample_rate = callerSampleRate;
72
+ } else if (needsRateInContentType) {
73
+ outputFormat.sample_rate = pcmDefault;
74
+ }
75
+ if (codec === "mp3" && modelOptions?.bit_rate !== void 0) {
76
+ outputFormat.bit_rate = modelOptions.bit_rate;
77
+ }
78
+ const sampleRateForContentType = callerSampleRate ?? pcmDefault;
79
+ const body = {
80
+ text,
81
+ voice_id: voice ?? "eve",
82
+ language: modelOptions?.language ?? "en",
83
+ output_format: outputFormat
84
+ };
85
+ if (modelOptions?.optimize_streaming_latency !== void 0) {
86
+ body.optimize_streaming_latency = modelOptions.optimize_streaming_latency;
87
+ }
88
+ if (modelOptions?.text_normalization !== void 0) {
89
+ body.text_normalization = modelOptions.text_normalization;
90
+ }
91
+ return { body, codec, sampleRateForContentType };
92
+ }
93
+ function pickCodec(codecOverride, format) {
94
+ if (codecOverride) return codecOverride;
95
+ if (!format) return "mp3";
96
+ switch (format) {
97
+ case "mp3":
98
+ case "wav":
99
+ case "pcm":
100
+ return format;
101
+ case "flac":
102
+ case "opus":
103
+ case "aac":
104
+ return "mp3";
105
+ default:
106
+ return "mp3";
107
+ }
108
+ }
109
+ function getContentType(codec, sampleRate) {
110
+ switch (codec) {
111
+ case "mp3":
112
+ return "audio/mpeg";
113
+ case "wav":
114
+ return "audio/wav";
115
+ case "pcm":
116
+ return `audio/L16;rate=${sampleRate}`;
117
+ case "mulaw":
118
+ return sampleRate === 8e3 ? "audio/basic" : `audio/PCMU;rate=${sampleRate}`;
119
+ case "alaw":
120
+ return sampleRate === 8e3 ? "audio/x-alaw-basic" : `audio/PCMA;rate=${sampleRate}`;
121
+ }
122
+ }
123
+ function createGrokSpeech(model, apiKey, config) {
124
+ return new GrokSpeechAdapter({ apiKey, ...config }, model);
125
+ }
126
+ function grokSpeech(model, config) {
127
+ const apiKey = getGrokApiKeyFromEnv();
128
+ return createGrokSpeech(model, apiKey, config);
129
+ }
130
+ export {
131
+ GrokSpeechAdapter,
132
+ buildTTSRequestBody,
133
+ createGrokSpeech,
134
+ getContentType,
135
+ grokSpeech
136
+ };
137
+ //# sourceMappingURL=tts.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"tts.js","sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport { arrayBufferToBase64, generateId, getGrokApiKeyFromEnv } from '../utils'\nimport type { TTSOptions, TTSResult } from '@tanstack/ai'\nimport type { GrokTTSModel } from '../model-meta'\nimport type {\n GrokTTSCodec,\n GrokTTSProviderOptions,\n GrokTTSVoice,\n} from '../audio/tts-provider-options'\n\nconst DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'\n\n/**\n * Configuration for the Grok TTS adapter.\n *\n * Unlike chat/image/summarize adapters, TTS does not use the OpenAI SDK\n * because xAI's `/v1/tts` endpoint is not OpenAI-compatible. This config\n * is a minimal subset suitable for direct `fetch` calls.\n */\nexport interface GrokSpeechConfig {\n apiKey: string\n baseURL?: string\n /** Additional headers to merge into every request (e.g., test IDs). */\n defaultHeaders?: Record<string, string>\n}\n\n/**\n * Grok Text-to-Speech Adapter.\n *\n * Talks to `POST {baseURL}/tts` per\n * https://docs.x.ai/developers/model-capabilities/audio/text-to-speech\n */\nexport class GrokSpeechAdapter<\n TModel extends GrokTTSModel,\n> extends BaseTTSAdapter<TModel, GrokTTSProviderOptions> {\n readonly name = 'grok' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n\n constructor(config: GrokSpeechConfig, model: TModel) {\n super(model, config)\n this.apiKey = config.apiKey\n this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\\/+$/, '')\n this.defaultHeaders = config.defaultHeaders ?? {}\n }\n\n async generateSpeech(\n options: TTSOptions<GrokTTSProviderOptions>,\n ): Promise<TTSResult> {\n const { logger } = options\n const { model, text, voice, format, modelOptions } = options\n\n logger.request(`activity=generateSpeech provider=grok model=${model}`, {\n provider: 'grok',\n model,\n })\n\n const { body, codec, sampleRateForContentType } = buildTTSRequestBody({\n text,\n voice,\n format,\n modelOptions,\n })\n\n try {\n const response = await fetch(`${this.baseURL}/tts`, {\n method: 'POST',\n headers: {\n // `defaultHeaders` first so the adapter's Authorization / Content-Type\n // always win — otherwise a caller-supplied `Authorization` header\n // could silently clobber the bearer token.\n ...this.defaultHeaders,\n Authorization: `Bearer ${this.apiKey}`,\n 'Content-Type': 'application/json',\n },\n body: JSON.stringify(body),\n })\n\n if (!response.ok) {\n const errorText = await response.text()\n throw new Error(\n `Grok TTS request failed: ${response.status} ${errorText}`,\n )\n }\n\n const arrayBuffer = await response.arrayBuffer()\n const audio = arrayBufferToBase64(arrayBuffer)\n\n return {\n id: generateId(this.name),\n model,\n audio,\n format: codec,\n contentType: getContentType(codec, sampleRateForContentType),\n }\n } catch (error) {\n logger.errors('grok.generateSpeech fatal', {\n error,\n source: 'grok.generateSpeech',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the JSON body for `POST /v1/tts`, resolving codec / sample-rate / voice\n * defaults in one place.\n *\n * Returns the request `body`, the resolved `codec`, and the `sampleRateForContentType`\n * used by the caller to label the response via `getContentType`.\n */\nexport function buildTTSRequestBody(options: {\n text: string\n voice: string | undefined\n format: TTSOptions['format'] | undefined\n modelOptions: GrokTTSProviderOptions | undefined\n}): {\n body: Record<string, unknown>\n codec: GrokTTSCodec\n sampleRateForContentType: number\n} {\n const { text, voice, format, modelOptions } = options\n\n const codec = pickCodec(modelOptions?.codec, format)\n\n // Only forward `sample_rate` when either:\n // - the caller explicitly set `modelOptions.sample_rate`, or\n // - the codec's Content-Type carries the rate (pcm → audio/L16;rate=…).\n // For mp3/wav/opus/aac/flac we leave sample_rate unset so xAI's server\n // default applies.\n const callerSampleRate = modelOptions?.sample_rate\n // Default sample rate documented in GrokTTSProviderOptions is 24000 Hz —\n // used only when we MUST attach a rate to the contentType (pcm) and the\n // caller didn't pick one.\n const pcmDefault = 24000\n const needsRateInContentType = codec === 'pcm'\n\n const outputFormat: Record<string, unknown> = { codec }\n if (callerSampleRate !== undefined) {\n outputFormat.sample_rate = callerSampleRate\n } else if (needsRateInContentType) {\n outputFormat.sample_rate = pcmDefault\n }\n if (codec === 'mp3' && modelOptions?.bit_rate !== undefined) {\n outputFormat.bit_rate = modelOptions.bit_rate\n }\n\n // pcm embeds the rate in `audio/L16;rate=…`; mulaw/alaw embed it in\n // `audio/PCMU;rate=…` / `audio/PCMA;rate=…` when non-default. mp3/wav\n // don't carry a rate parameter so the value is unused for those.\n const sampleRateForContentType = callerSampleRate ?? pcmDefault\n\n const body: Record<string, unknown> = {\n text,\n voice_id: (voice as GrokTTSVoice | undefined) ?? 'eve',\n language: modelOptions?.language ?? 'en',\n output_format: outputFormat,\n }\n if (modelOptions?.optimize_streaming_latency !== undefined) {\n body.optimize_streaming_latency = modelOptions.optimize_streaming_latency\n }\n if (modelOptions?.text_normalization !== undefined) {\n body.text_normalization = modelOptions.text_normalization\n }\n\n return { body, codec, sampleRateForContentType }\n}\n\n/**\n * Maps the cross-provider `TTSOptions.format` onto Grok's supported codecs.\n * `opus`, `aac`, and `flac` are not supported by xAI TTS (which only exposes\n * mp3/wav/pcm/mulaw/alaw) — we fall back to mp3. An explicit\n * `modelOptions.codec` always wins.\n */\nfunction pickCodec(\n codecOverride: GrokTTSCodec | undefined,\n format: TTSOptions['format'] | undefined,\n): GrokTTSCodec {\n if (codecOverride) return codecOverride\n if (!format) return 'mp3'\n switch (format) {\n case 'mp3':\n case 'wav':\n case 'pcm':\n return format\n case 'flac':\n case 'opus':\n case 'aac':\n return 'mp3'\n default:\n return 'mp3'\n }\n}\n\nexport function getContentType(\n codec: GrokTTSCodec,\n sampleRate: number,\n): string {\n switch (codec) {\n case 'mp3':\n return 'audio/mpeg'\n case 'wav':\n return 'audio/wav'\n case 'pcm':\n // `audio/L16` requires a `rate` parameter per RFC 3551/3555.\n return `audio/L16;rate=${sampleRate}`\n case 'mulaw':\n // `audio/basic` is 8 kHz mono by RFC 2046 registration. For non-8kHz\n // streams xAI still produces mulaw-encoded bytes at the requested\n // rate, but the registered MIME can't carry that rate — so we use\n // the non-standard but commonly-supported `audio/PCMU;rate=…` (RFC 3551\n // RTP payload name) whenever the caller asked for a rate other than\n // 8000, and keep `audio/basic` for the standard 8kHz case.\n return sampleRate === 8000\n ? 'audio/basic'\n : `audio/PCMU;rate=${sampleRate}`\n case 'alaw':\n return sampleRate === 8000\n ? 'audio/x-alaw-basic'\n : `audio/PCMA;rate=${sampleRate}`\n }\n}\n\n/**\n * Creates a Grok speech (TTS) adapter with an explicit API key.\n *\n * @example\n * ```typescript\n * const adapter = createGrokSpeech('grok-tts', 'xai-...')\n * const result = await generateSpeech({\n * adapter,\n * text: 'Hello from Grok',\n * voice: 'eve',\n * })\n * ```\n */\nexport function createGrokSpeech<TModel extends GrokTTSModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GrokSpeechConfig, 'apiKey'>,\n): GrokSpeechAdapter<TModel> {\n return new GrokSpeechAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Grok speech (TTS) adapter, reading the API key from\n * `XAI_API_KEY` in the environment.\n *\n * @throws Error if `XAI_API_KEY` is not set.\n */\nexport function grokSpeech<TModel extends GrokTTSModel>(\n model: TModel,\n config?: Omit<GrokSpeechConfig, 'apiKey'>,\n): GrokSpeechAdapter<TModel> {\n const apiKey = getGrokApiKeyFromEnv()\n return createGrokSpeech(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;AAUA,MAAM,wBAAwB;AAsBvB,MAAM,0BAEH,eAA+C;AAAA,EAOvD,YAAY,QAA0B,OAAe;AACnD,UAAM,OAAO,MAAM;AAPrB,SAAS,OAAO;AAQd,SAAK,SAAS,OAAO;AACrB,SAAK,WAAW,OAAO,WAAW,uBAAuB,QAAQ,QAAQ,EAAE;AAC3E,SAAK,iBAAiB,OAAO,kBAAkB,CAAA;AAAA,EACjD;AAAA,EAEA,MAAM,eACJ,SACoB;AACpB,UAAM,EAAE,WAAW;AACnB,UAAM,EAAE,OAAO,MAAM,OAAO,QAAQ,iBAAiB;AAErD,WAAO,QAAQ,+CAA+C,KAAK,IAAI;AAAA,MACrE,UAAU;AAAA,MACV;AAAA,IAAA,CACD;AAED,UAAM,EAAE,MAAM,OAAO,yBAAA,IAA6B,oBAAoB;AAAA,MACpE;AAAA,MACA;AAAA,MACA;AAAA,MACA;AAAA,IAAA,CACD;AAED,QAAI;AACF,YAAM,WAAW,MAAM,MAAM,GAAG,KAAK,OAAO,QAAQ;AAAA,QAClD,QAAQ;AAAA,QACR,SAAS;AAAA;AAAA;AAAA;AAAA,UAIP,GAAG,KAAK;AAAA,UACR,eAAe,UAAU,KAAK,MAAM;AAAA,UACpC,gBAAgB;AAAA,QAAA;AAAA,QAElB,MAAM,KAAK,UAAU,IAAI;AAAA,MAAA,CAC1B;AAED,UAAI,CAAC,SAAS,IAAI;AAChB,cAAM,YAAY,MAAM,SAAS,KAAA;AACjC,cAAM,IAAI;AAAA,UACR,4BAA4B,SAAS,MAAM,IAAI,SAAS;AAAA,QAAA;AAAA,MAE5D;AAEA,YAAM,cAAc,MAAM,SAAS,YAAA;AACnC,YAAM,QAAQ,oBAAoB,WAAW;AAE7C,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA;AAAA,QACA,QAAQ;AAAA,QACR,aAAa,eAAe,OAAO,wBAAwB;AAAA,MAAA;AAAA,IAE/D,SAAS,OAAO;AACd,aAAO,OAAO,6BAA6B;AAAA,QACzC;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AASO,SAAS,oBAAoB,SASlC;AACA,QAAM,EAAE,MAAM,OAAO,QAAQ,iBAAiB;AAE9C,QAAM,QAAQ,UAAU,cAAc,OAAO,MAAM;AAOnD,QAAM,mBAAmB,cAAc;AAIvC,QAAM,aAAa;AACnB,QAAM,yBAAyB,UAAU;AAEzC,QAAM,eAAwC,EAAE,MAAA;AAChD,MAAI,qBAAqB,QAAW;AAClC,iBAAa,cAAc;AAAA,EAC7B,WAAW,wBAAwB;AACjC,iBAAa,cAAc;AAAA,EAC7B;AACA,MAAI,UAAU,SAAS,cAAc,aAAa,QAAW;AAC3D,iBAAa,WAAW,aAAa;AAAA,EACvC;AAKA,QAAM,2BAA2B,oBAAoB;AAErD,QAAM,OAAgC;AAAA,IACpC;AAAA,IACA,UAAW,SAAsC;AAAA,IACjD,UAAU,cAAc,YAAY;AAAA,IACpC,eAAe;AAAA,EAAA;AAEjB,MAAI,cAAc,+BAA+B,QAAW;AAC1D,SAAK,6BAA6B,aAAa;AAAA,EACjD;AACA,MAAI,cAAc,uBAAuB,QAAW;AAClD,SAAK,qBAAqB,aAAa;AAAA,EACzC;AAEA,SAAO,EAAE,MAAM,OAAO,yBAAA;AACxB;AAQA,SAAS,UACP,eACA,QACc;AACd,MAAI,cAAe,QAAO;AAC1B,MAAI,CAAC,OAAQ,QAAO;AACpB,UAAQ,QAAA;AAAA,IACN,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AACH,aAAO;AAAA,IACT;AACE,aAAO;AAAA,EAAA;AAEb;AAEO,SAAS,eACd,OACA,YACQ;AACR,UAAQ,OAAA;AAAA,IACN,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AAEH,aAAO,kBAAkB,UAAU;AAAA,IACrC,KAAK;AAOH,aAAO,eAAe,MAClB,gBACA,mBAAmB,UAAU;AAAA,IACnC,KAAK;AACH,aAAO,eAAe,MAClB,uBACA,mBAAmB,UAAU;AAAA,EAAA;AAEvC;AAeO,SAAS,iBACd,OACA,QACA,QAC2B;AAC3B,SAAO,IAAI,kBAAkB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAC3D;AAQO,SAAS,WACd,OACA,QAC2B;AAC3B,QAAM,SAAS,qBAAA;AACf,SAAO,iBAAiB,OAAO,QAAQ,MAAM;AAC/C;"}
@@ -0,0 +1,41 @@
1
+ /**
2
+ * Grok STT supported audio formats.
3
+ * See https://docs.x.ai/developers/rest-api-reference/inference/voice
4
+ */
5
+ export type GrokSTTAudioFormat = 'pcm' | 'mulaw' | 'alaw' | 'wav' | 'mp3' | 'ogg' | 'opus' | 'flac' | 'aac' | 'mp4' | 'm4a' | 'mkv';
6
+ /**
7
+ * Provider-specific options for Grok transcription (`POST /v1/stt`).
8
+ */
9
+ export interface GrokTranscriptionProviderOptions {
10
+ /**
11
+ * The format of the provided audio. Required for raw codecs (pcm, mulaw, alaw).
12
+ */
13
+ audio_format?: GrokSTTAudioFormat;
14
+ /**
15
+ * Sample rate of the audio (Hz). Required for raw codecs.
16
+ */
17
+ sample_rate?: number;
18
+ /**
19
+ * Apply inverse text normalization (e.g. "one hundred" → "100"). Requires
20
+ * `language` to be set on the core `TranscriptionOptions`.
21
+ *
22
+ * NOTE: xAI's STT API exposes this on the wire as `format` (a boolean
23
+ * toggle). We surface it under the clearer name
24
+ * `inverse_text_normalization` on the SDK, and translate to the wire name
25
+ * inside the adapter.
26
+ */
27
+ inverse_text_normalization?: boolean;
28
+ /**
29
+ * Treat the audio as multichannel. When enabled, `channels` must also be set.
30
+ */
31
+ multichannel?: boolean;
32
+ /**
33
+ * Channel count for multichannel raw audio (2–8).
34
+ */
35
+ channels?: number;
36
+ /**
37
+ * Enable speaker diarization. When true, response words include a `speaker`
38
+ * field.
39
+ */
40
+ diarize?: boolean;
41
+ }
@@ -0,0 +1,42 @@
1
+ /**
2
+ * Grok TTS voice options.
3
+ * See https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
4
+ */
5
+ export type GrokTTSVoice = 'eve' | 'ara' | 'rex' | 'sal' | 'leo';
6
+ /**
7
+ * Grok TTS output audio codecs.
8
+ * Grok does NOT support opus or aac; those formats are mapped to mp3.
9
+ */
10
+ export type GrokTTSCodec = 'mp3' | 'wav' | 'pcm' | 'mulaw' | 'alaw';
11
+ /**
12
+ * Provider-specific options for Grok TTS (`POST /v1/tts`).
13
+ */
14
+ export interface GrokTTSProviderOptions {
15
+ /**
16
+ * BCP-47 language code (e.g., `en`, `zh`, `pt-BR`) or `'auto'` for detection.
17
+ * Defaults to `'en'` when not provided.
18
+ */
19
+ language?: string;
20
+ /**
21
+ * Audio codec. Overrides the `format` field on `TTSOptions` when set.
22
+ */
23
+ codec?: GrokTTSCodec;
24
+ /**
25
+ * Sample rate in Hz. Valid values: 8000, 16000, 22050, 24000, 44100, 48000.
26
+ * Defaults to 24000.
27
+ */
28
+ sample_rate?: 8000 | 16000 | 22050 | 24000 | 44100 | 48000;
29
+ /**
30
+ * Bit rate for MP3 output. Ignored for other codecs.
31
+ * Valid values: 32000, 64000, 96000, 128000, 192000. Defaults to 128000.
32
+ */
33
+ bit_rate?: 32000 | 64000 | 96000 | 128000 | 192000;
34
+ /**
35
+ * Set to 1 for lower latency streaming; 0 (default) for normal quality.
36
+ */
37
+ optimize_streaming_latency?: 0 | 1;
38
+ /**
39
+ * Enable text normalization. Defaults to false.
40
+ */
41
+ text_normalization?: boolean;
42
+ }
@@ -2,6 +2,12 @@ export { GrokTextAdapter, createGrokText, grokText, type GrokTextConfig, type Gr
2
2
  export { GrokSummarizeAdapter, createGrokSummarize, grokSummarize, type GrokSummarizeConfig, type GrokSummarizeProviderOptions, type GrokSummarizeModel, } from './adapters/summarize.js';
3
3
  export { GrokImageAdapter, createGrokImage, grokImage, type GrokImageConfig, } from './adapters/image.js';
4
4
  export type { GrokImageProviderOptions, GrokImageModelProviderOptionsByName, } from './image/image-provider-options.js';
5
- export type { GrokChatModelProviderOptionsByName, GrokChatModelToolCapabilitiesByName, GrokModelInputModalitiesByName, ResolveProviderOptions, ResolveInputModalities, GrokChatModel, GrokImageModel, } from './model-meta.js';
6
- export { GROK_CHAT_MODELS, GROK_IMAGE_MODELS } from './model-meta.js';
5
+ export { GrokSpeechAdapter, createGrokSpeech, grokSpeech, type GrokSpeechConfig, } from './adapters/tts.js';
6
+ export type { GrokTTSProviderOptions, GrokTTSVoice, GrokTTSCodec, } from './audio/tts-provider-options.js';
7
+ export { GrokTranscriptionAdapter, createGrokTranscription, grokTranscription, type GrokTranscriptionConfig, } from './adapters/transcription.js';
8
+ export type { GrokTranscriptionProviderOptions, GrokSTTAudioFormat, } from './audio/transcription-provider-options.js';
9
+ export type { GrokChatModelProviderOptionsByName, GrokChatModelToolCapabilitiesByName, GrokModelInputModalitiesByName, ResolveProviderOptions, ResolveInputModalities, GrokChatModel, GrokImageModel, GrokTTSModel, GrokTranscriptionModel, GrokRealtimeModel, } from './model-meta.js';
10
+ export { GROK_CHAT_MODELS, GROK_IMAGE_MODELS, GROK_TTS_MODELS, GROK_TRANSCRIPTION_MODELS, GROK_REALTIME_MODELS, } from './model-meta.js';
7
11
  export type { GrokTextMetadata, GrokImageMetadata, GrokAudioMetadata, GrokVideoMetadata, GrokDocumentMetadata, GrokMessageMetadataByModality, } from './message-types.js';
12
+ export { grokRealtimeToken, grokRealtime } from './realtime/index.js';
13
+ export type { GrokRealtimeVoice, GrokRealtimeTokenOptions, GrokRealtimeOptions, GrokTurnDetection, GrokSemanticVADConfig, GrokServerVADConfig, } from './realtime/index.js';
package/dist/esm/index.js CHANGED
@@ -1,18 +1,33 @@
1
1
  import { GrokTextAdapter, createGrokText, grokText } from "./adapters/text.js";
2
2
  import { GrokSummarizeAdapter, createGrokSummarize, grokSummarize } from "./adapters/summarize.js";
3
3
  import { GrokImageAdapter, createGrokImage, grokImage } from "./adapters/image.js";
4
- import { GROK_CHAT_MODELS, GROK_IMAGE_MODELS } from "./model-meta.js";
4
+ import { GrokSpeechAdapter, createGrokSpeech, grokSpeech } from "./adapters/tts.js";
5
+ import { GrokTranscriptionAdapter, createGrokTranscription, grokTranscription } from "./adapters/transcription.js";
6
+ import { GROK_CHAT_MODELS, GROK_IMAGE_MODELS, GROK_REALTIME_MODELS, GROK_TRANSCRIPTION_MODELS, GROK_TTS_MODELS } from "./model-meta.js";
7
+ import { grokRealtimeToken } from "./realtime/token.js";
8
+ import { grokRealtime } from "./realtime/adapter.js";
5
9
  export {
6
10
  GROK_CHAT_MODELS,
7
11
  GROK_IMAGE_MODELS,
12
+ GROK_REALTIME_MODELS,
13
+ GROK_TRANSCRIPTION_MODELS,
14
+ GROK_TTS_MODELS,
8
15
  GrokImageAdapter,
16
+ GrokSpeechAdapter,
9
17
  GrokSummarizeAdapter,
10
18
  GrokTextAdapter,
19
+ GrokTranscriptionAdapter,
11
20
  createGrokImage,
21
+ createGrokSpeech,
12
22
  createGrokSummarize,
13
23
  createGrokText,
24
+ createGrokTranscription,
14
25
  grokImage,
26
+ grokRealtime,
27
+ grokRealtimeToken,
28
+ grokSpeech,
15
29
  grokSummarize,
16
- grokText
30
+ grokText,
31
+ grokTranscription
17
32
  };
18
33
  //# sourceMappingURL=index.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;"}
1
+ {"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;"}
@@ -215,8 +215,14 @@ export declare const GROK_CHAT_MODELS: readonly ["grok-4-1-fast-reasoning", "gro
215
215
  * Grok Image Generation Models
216
216
  */
217
217
  export declare const GROK_IMAGE_MODELS: readonly ["grok-2-image-1212"];
218
+ export declare const GROK_TTS_MODELS: readonly ["grok-tts"];
219
+ export declare const GROK_TRANSCRIPTION_MODELS: readonly ["grok-stt"];
220
+ export declare const GROK_REALTIME_MODELS: readonly ["grok-voice-fast-1.0", "grok-voice-think-fast-1.0"];
218
221
  export type GrokChatModel = (typeof GROK_CHAT_MODELS)[number];
219
222
  export type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number];
223
+ export type GrokTTSModel = (typeof GROK_TTS_MODELS)[number];
224
+ export type GrokTranscriptionModel = (typeof GROK_TRANSCRIPTION_MODELS)[number];
225
+ export type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number];
220
226
  /**
221
227
  * Type-only map from Grok chat model name to its supported input modalities.
222
228
  * Used for type inference when constructing multimodal messages.
@@ -48,8 +48,29 @@ const GROK_CHAT_MODELS = [
48
48
  GROK_4_20_MULTI_AGENT.name
49
49
  ];
50
50
  const GROK_IMAGE_MODELS = [GROK_2_IMAGE.name];
51
+ const GROK_TTS = {
52
+ name: "grok-tts"
53
+ };
54
+ const GROK_STT = {
55
+ name: "grok-stt"
56
+ };
57
+ const GROK_VOICE_FAST_1 = {
58
+ name: "grok-voice-fast-1.0"
59
+ };
60
+ const GROK_VOICE_THINK_FAST_1 = {
61
+ name: "grok-voice-think-fast-1.0"
62
+ };
63
+ const GROK_TTS_MODELS = [GROK_TTS.name];
64
+ const GROK_TRANSCRIPTION_MODELS = [GROK_STT.name];
65
+ const GROK_REALTIME_MODELS = [
66
+ GROK_VOICE_FAST_1.name,
67
+ GROK_VOICE_THINK_FAST_1.name
68
+ ];
51
69
  export {
52
70
  GROK_CHAT_MODELS,
53
- GROK_IMAGE_MODELS
71
+ GROK_IMAGE_MODELS,
72
+ GROK_REALTIME_MODELS,
73
+ GROK_TRANSCRIPTION_MODELS,
74
+ GROK_TTS_MODELS
54
75
  };
55
76
  //# sourceMappingURL=model-meta.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"model-meta.js","sources":["../../src/model-meta.ts"],"sourcesContent":["/**\n * Model metadata interface for documentation and type inference\n */\ninterface ModelMeta {\n name: string\n supports: {\n input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>\n output: Array<'text' | 'image' | 'audio' | 'video'>\n capabilities?: Array<'reasoning' | 'tool_calling' | 'structured_outputs'>\n tools?: ReadonlyArray<never>\n }\n max_input_tokens?: number\n max_output_tokens?: number\n context_window?: number\n knowledge_cutoff?: string\n pricing?: {\n input: {\n normal: number\n cached?: number\n }\n output: {\n normal: number\n }\n }\n}\n\nconst GROK_4_1_FAST_REASONING = {\n name: 'grok-4-1-fast-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_1_FAST_NON_REASONING = {\n name: 'grok-4-1-fast-non-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_CODE_FAST_1 = {\n name: 'grok-code-fast-1',\n context_window: 256_000,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.02,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_FAST_REASONING = {\n name: 'grok-4-fast-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_FAST_NON_REASONING = {\n name: 'grok-4-fast-non-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4 = {\n name: 'grok-4',\n context_window: 256_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 3,\n cached: 0.75,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_3_MINI = {\n name: 'grok-3-mini',\n context_window: 131_072,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.3,\n cached: 0.075,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_3 = {\n name: 'grok-3',\n context_window: 131_072,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 3,\n cached: 0.75,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_2_VISION = {\n name: 'grok-2-vision-1212',\n context_window: 32_768,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n },\n output: {\n normal: 10,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_2_IMAGE = {\n name: 'grok-2-image-1212',\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0.07,\n },\n output: {\n normal: 0.07,\n },\n },\n} as const satisfies ModelMeta\n\n/**\n * Grok Chat Models\n * Based on xAI's available models as of 2025\n */\nconst GROK_4_20 = {\n name: 'grok-4.20',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image', 'document'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n cached: 0.2,\n },\n output: {\n normal: 6,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_20_MULTI_AGENT = {\n name: 'grok-4.20-multi-agent',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image', 'document'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n cached: 0.2,\n },\n output: {\n normal: 6,\n },\n },\n} as const satisfies ModelMeta\n\nexport const GROK_CHAT_MODELS = [\n GROK_4_1_FAST_REASONING.name,\n GROK_4_1_FAST_NON_REASONING.name,\n GROK_CODE_FAST_1.name,\n GROK_4_FAST_REASONING.name,\n GROK_4_FAST_NON_REASONING.name,\n GROK_4.name,\n GROK_3.name,\n GROK_3_MINI.name,\n GROK_2_VISION.name,\n\n GROK_4_20.name,\n GROK_4_20_MULTI_AGENT.name,\n] as const\n\n/**\n * Grok Image Generation Models\n */\nexport const GROK_IMAGE_MODELS = [GROK_2_IMAGE.name] as const\n\nexport type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]\nexport type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]\n\n/**\n * Type-only map from Grok chat model name to its supported input modalities.\n * Used for type inference when constructing multimodal messages.\n */\nexport type GrokModelInputModalitiesByName = {\n [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.input\n [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.input\n [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.input\n [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.input\n [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.input\n [GROK_4.name]: typeof GROK_4.supports.input\n [GROK_3.name]: typeof GROK_3.supports.input\n [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.input\n [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.input\n [GROK_4_20.name]: typeof GROK_4_20.supports.input\n [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.input\n}\n\n/**\n * Type-only map from Grok chat model name to its provider options type.\n * Since Grok uses OpenAI-compatible API, we reuse OpenAI provider options.\n */\nexport type GrokChatModelProviderOptionsByName = {\n [K in (typeof GROK_CHAT_MODELS)[number]]: GrokProviderOptions\n}\n\n/**\n * Type-only map from Grok chat model name to its supported provider tools.\n * Grok exposes no provider-specific tool factories, so every model gets an\n * empty tuple. This ensures that passing an Anthropic/OpenAI ProviderTool to\n * a Grok adapter produces a compile-time type error.\n */\nexport type GrokChatModelToolCapabilitiesByName = {\n [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.tools\n [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.tools\n [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.tools\n [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.tools\n [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.tools\n [GROK_4.name]: typeof GROK_4.supports.tools\n [GROK_3.name]: typeof GROK_3.supports.tools\n [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.tools\n [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.tools\n [GROK_4_20.name]: typeof GROK_4_20.supports.tools\n [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.tools\n}\n\n/**\n * Grok-specific provider options\n * Based on OpenAI-compatible API options\n */\nexport interface GrokProviderOptions {\n /** Temperature for response generation (0-2) */\n temperature?: number\n /** Maximum tokens in the response */\n max_tokens?: number\n /** Top-p sampling parameter */\n top_p?: number\n /** Frequency penalty (-2.0 to 2.0) */\n frequency_penalty?: number\n /** Presence penalty (-2.0 to 2.0) */\n presence_penalty?: number\n /** Stop sequences */\n stop?: string | Array<string>\n /** A unique identifier representing your end-user */\n user?: string\n}\n\n// ===========================\n// Type Resolution Helpers\n// ===========================\n\n/**\n * Resolve provider options for a specific model.\n * If the model has explicit options in the map, use those; otherwise use base options.\n */\nexport type ResolveProviderOptions<TModel extends string> =\n TModel extends keyof GrokChatModelProviderOptionsByName\n ? GrokChatModelProviderOptionsByName[TModel]\n : GrokProviderOptions\n\n/**\n * Resolve input modalities for a specific model.\n * If the model has explicit modalities in the map, use those; otherwise use text only.\n */\nexport type ResolveInputModalities<TModel extends string> =\n TModel extends keyof GrokModelInputModalitiesByName\n ? GrokModelInputModalitiesByName[TModel]\n : readonly ['text']\n"],"names":[],"mappings":"AA0BA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAiBR;AAEA,MAAM,8BAA8B;AAAA,EAClC,MAAM;AAiBR;AAEA,MAAM,mBAAmB;AAAA,EACvB,MAAM;AAiBR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAiBR;AAEA,MAAM,4BAA4B;AAAA,EAChC,MAAM;AAiBR;AAEA,MAAM,SAAS;AAAA,EACb,MAAM;AAiBR;AAEA,MAAM,cAAc;AAAA,EAClB,MAAM;AAiBR;AAEA,MAAM,SAAS;AAAA,EACb,MAAM;AAiBR;AAEA,MAAM,gBAAgB;AAAA,EACpB,MAAM;AAgBR;AAEA,MAAM,eAAe;AAAA,EACnB,MAAM;AAaR;AAMA,MAAM,YAAY;AAAA,EAChB,MAAM;AAiBR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAiBR;AAEO,MAAM,mBAAmB;AAAA,EAC9B,wBAAwB;AAAA,EACxB,4BAA4B;AAAA,EAC5B,iBAAiB;AAAA,EACjB,sBAAsB;AAAA,EACtB,0BAA0B;AAAA,EAC1B,OAAO;AAAA,EACP,OAAO;AAAA,EACP,YAAY;AAAA,EACZ,cAAc;AAAA,EAEd,UAAU;AAAA,EACV,sBAAsB;AACxB;AAKO,MAAM,oBAAoB,CAAC,aAAa,IAAI;"}
1
+ {"version":3,"file":"model-meta.js","sources":["../../src/model-meta.ts"],"sourcesContent":["/**\n * Model metadata interface for documentation and type inference\n */\ninterface ModelMeta {\n name: string\n supports: {\n input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>\n output: Array<'text' | 'image' | 'audio' | 'video'>\n capabilities?: Array<'reasoning' | 'tool_calling' | 'structured_outputs'>\n tools?: ReadonlyArray<never>\n }\n max_input_tokens?: number\n max_output_tokens?: number\n context_window?: number\n knowledge_cutoff?: string\n pricing?: {\n input: {\n normal: number\n cached?: number\n }\n output: {\n normal: number\n }\n }\n}\n\nconst GROK_4_1_FAST_REASONING = {\n name: 'grok-4-1-fast-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_1_FAST_NON_REASONING = {\n name: 'grok-4-1-fast-non-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_CODE_FAST_1 = {\n name: 'grok-code-fast-1',\n context_window: 256_000,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.02,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_FAST_REASONING = {\n name: 'grok-4-fast-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_FAST_NON_REASONING = {\n name: 'grok-4-fast-non-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4 = {\n name: 'grok-4',\n context_window: 256_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 3,\n cached: 0.75,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_3_MINI = {\n name: 'grok-3-mini',\n context_window: 131_072,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.3,\n cached: 0.075,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_3 = {\n name: 'grok-3',\n context_window: 131_072,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 3,\n cached: 0.75,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_2_VISION = {\n name: 'grok-2-vision-1212',\n context_window: 32_768,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n },\n output: {\n normal: 10,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_2_IMAGE = {\n name: 'grok-2-image-1212',\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0.07,\n },\n output: {\n normal: 0.07,\n },\n },\n} as const satisfies ModelMeta\n\n/**\n * Grok Chat Models\n * Based on xAI's available models as of 2025\n */\nconst GROK_4_20 = {\n name: 'grok-4.20',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image', 'document'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n cached: 0.2,\n },\n output: {\n normal: 6,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_20_MULTI_AGENT = {\n name: 'grok-4.20-multi-agent',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image', 'document'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n cached: 0.2,\n },\n output: {\n normal: 6,\n },\n },\n} as const satisfies ModelMeta\n\nexport const GROK_CHAT_MODELS = [\n GROK_4_1_FAST_REASONING.name,\n GROK_4_1_FAST_NON_REASONING.name,\n GROK_CODE_FAST_1.name,\n GROK_4_FAST_REASONING.name,\n GROK_4_FAST_NON_REASONING.name,\n GROK_4.name,\n GROK_3.name,\n GROK_3_MINI.name,\n GROK_2_VISION.name,\n\n GROK_4_20.name,\n GROK_4_20_MULTI_AGENT.name,\n] as const\n\n/**\n * Grok Image Generation Models\n */\nexport const GROK_IMAGE_MODELS = [GROK_2_IMAGE.name] as const\n\n// xAI's `/v1/tts` endpoint is endpoint-addressed and does not take a `model`\n// parameter. This synthetic identifier satisfies the SDK's `TTSOptions.model`\n// contract and provides a stable value for logging and fixture matching.\nconst GROK_TTS = {\n name: 'grok-tts',\n supports: {\n input: ['text'],\n output: ['audio'],\n },\n} as const satisfies ModelMeta\n\n// xAI's `/v1/stt` endpoint is endpoint-addressed and does not take a `model`\n// parameter. This synthetic identifier satisfies the SDK's\n// `TranscriptionOptions.model` contract.\nconst GROK_STT = {\n name: 'grok-stt',\n supports: {\n input: ['audio'],\n output: ['text'],\n },\n} as const satisfies ModelMeta\n\nconst GROK_VOICE_FAST_1 = {\n name: 'grok-voice-fast-1.0',\n supports: {\n input: ['audio', 'text'],\n output: ['audio', 'text'],\n capabilities: ['tool_calling'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta\n\nconst GROK_VOICE_THINK_FAST_1 = {\n name: 'grok-voice-think-fast-1.0',\n supports: {\n input: ['audio', 'text'],\n output: ['audio', 'text'],\n capabilities: ['reasoning', 'tool_calling'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta\n\nexport const GROK_TTS_MODELS = [GROK_TTS.name] as const\n\nexport const GROK_TRANSCRIPTION_MODELS = [GROK_STT.name] as const\n\nexport const GROK_REALTIME_MODELS = [\n GROK_VOICE_FAST_1.name,\n GROK_VOICE_THINK_FAST_1.name,\n] as const\n\nexport type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]\nexport type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]\nexport type GrokTTSModel = (typeof GROK_TTS_MODELS)[number]\nexport type GrokTranscriptionModel = (typeof GROK_TRANSCRIPTION_MODELS)[number]\nexport type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]\n\n/**\n * Type-only map from Grok chat model name to its supported input modalities.\n * Used for type inference when constructing multimodal messages.\n */\nexport type GrokModelInputModalitiesByName = {\n [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.input\n [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.input\n [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.input\n [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.input\n [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.input\n [GROK_4.name]: typeof GROK_4.supports.input\n [GROK_3.name]: typeof GROK_3.supports.input\n [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.input\n [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.input\n [GROK_4_20.name]: typeof GROK_4_20.supports.input\n [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.input\n}\n\n/**\n * Type-only map from Grok chat model name to its provider options type.\n * Since Grok uses OpenAI-compatible API, we reuse OpenAI provider options.\n */\nexport type GrokChatModelProviderOptionsByName = {\n [K in (typeof GROK_CHAT_MODELS)[number]]: GrokProviderOptions\n}\n\n/**\n * Type-only map from Grok chat model name to its supported provider tools.\n * Grok exposes no provider-specific tool factories, so every model gets an\n * empty tuple. This ensures that passing an Anthropic/OpenAI ProviderTool to\n * a Grok adapter produces a compile-time type error.\n */\nexport type GrokChatModelToolCapabilitiesByName = {\n [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.tools\n [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.tools\n [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.tools\n [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.tools\n [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.tools\n [GROK_4.name]: typeof GROK_4.supports.tools\n [GROK_3.name]: typeof GROK_3.supports.tools\n [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.tools\n [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.tools\n [GROK_4_20.name]: typeof GROK_4_20.supports.tools\n [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.tools\n}\n\n/**\n * Grok-specific provider options\n * Based on OpenAI-compatible API options\n */\nexport interface GrokProviderOptions {\n /** Temperature for response generation (0-2) */\n temperature?: number\n /** Maximum tokens in the response */\n max_tokens?: number\n /** Top-p sampling parameter */\n top_p?: number\n /** Frequency penalty (-2.0 to 2.0) */\n frequency_penalty?: number\n /** Presence penalty (-2.0 to 2.0) */\n presence_penalty?: number\n /** Stop sequences */\n stop?: string | Array<string>\n /** A unique identifier representing your end-user */\n user?: string\n}\n\n// ===========================\n// Type Resolution Helpers\n// ===========================\n\n/**\n * Resolve provider options for a specific model.\n * If the model has explicit options in the map, use those; otherwise use base options.\n */\nexport type ResolveProviderOptions<TModel extends string> =\n TModel extends keyof GrokChatModelProviderOptionsByName\n ? GrokChatModelProviderOptionsByName[TModel]\n : GrokProviderOptions\n\n/**\n * Resolve input modalities for a specific model.\n * If the model has explicit modalities in the map, use those; otherwise use text only.\n */\nexport type ResolveInputModalities<TModel extends string> =\n TModel extends keyof GrokModelInputModalitiesByName\n ? GrokModelInputModalitiesByName[TModel]\n : readonly ['text']\n"],"names":[],"mappings":"AA0BA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAiBR;AAEA,MAAM,8BAA8B;AAAA,EAClC,MAAM;AAiBR;AAEA,MAAM,mBAAmB;AAAA,EACvB,MAAM;AAiBR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAiBR;AAEA,MAAM,4BAA4B;AAAA,EAChC,MAAM;AAiBR;AAEA,MAAM,SAAS;AAAA,EACb,MAAM;AAiBR;AAEA,MAAM,cAAc;AAAA,EAClB,MAAM;AAiBR;AAEA,MAAM,SAAS;AAAA,EACb,MAAM;AAiBR;AAEA,MAAM,gBAAgB;AAAA,EACpB,MAAM;AAgBR;AAEA,MAAM,eAAe;AAAA,EACnB,MAAM;AAaR;AAMA,MAAM,YAAY;AAAA,EAChB,MAAM;AAiBR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAiBR;AAEO,MAAM,mBAAmB;AAAA,EAC9B,wBAAwB;AAAA,EACxB,4BAA4B;AAAA,EAC5B,iBAAiB;AAAA,EACjB,sBAAsB;AAAA,EACtB,0BAA0B;AAAA,EAC1B,OAAO;AAAA,EACP,OAAO;AAAA,EACP,YAAY;AAAA,EACZ,cAAc;AAAA,EAEd,UAAU;AAAA,EACV,sBAAsB;AACxB;AAKO,MAAM,oBAAoB,CAAC,aAAa,IAAI;AAKnD,MAAM,WAAW;AAAA,EACf,MAAM;AAKR;AAKA,MAAM,WAAW;AAAA,EACf,MAAM;AAKR;AAEA,MAAM,oBAAoB;AAAA,EACxB,MAAM;AAOR;AAEA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAOR;AAEO,MAAM,kBAAkB,CAAC,SAAS,IAAI;AAEtC,MAAM,4BAA4B,CAAC,SAAS,IAAI;AAEhD,MAAM,uBAAuB;AAAA,EAClC,kBAAkB;AAAA,EAClB,wBAAwB;AAC1B;"}
@@ -0,0 +1,21 @@
1
+ import { RealtimeAdapter } from './realtime-contract.js';
2
+ import { GrokRealtimeOptions } from './types.js';
3
+ /**
4
+ * Creates a Grok realtime adapter for client-side use.
5
+ *
6
+ * Uses WebRTC for browser connections (default). Mirrors the OpenAI realtime
7
+ * adapter because xAI's Voice Agent API is OpenAI-realtime-compatible — the
8
+ * only differences are the endpoint URL and default model.
9
+ *
10
+ * @example
11
+ * ```typescript
12
+ * import { RealtimeClient } from '@tanstack/ai-client'
13
+ * import { grokRealtime } from '@tanstack/ai-grok'
14
+ *
15
+ * const client = new RealtimeClient({
16
+ * getToken: () => fetch('/api/realtime-token').then(r => r.json()),
17
+ * adapter: grokRealtime(),
18
+ * })
19
+ * ```
20
+ */
21
+ export declare function grokRealtime(options?: GrokRealtimeOptions): RealtimeAdapter;