@tanstack/ai-grok 0.14.1 → 0.14.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +4 -4
- package/dist/esm/adapters/image.d.ts +0 -89
- package/dist/esm/adapters/image.js +0 -205
- package/dist/esm/adapters/image.js.map +0 -1
- package/dist/esm/adapters/summarize.d.ts +0 -51
- package/dist/esm/adapters/summarize.js +0 -20
- package/dist/esm/adapters/summarize.js.map +0 -1
- package/dist/esm/adapters/text.d.ts +0 -76
- package/dist/esm/adapters/text.js +0 -44
- package/dist/esm/adapters/text.js.map +0 -1
- package/dist/esm/adapters/transcription.d.ts +0 -84
- package/dist/esm/adapters/transcription.js +0 -122
- package/dist/esm/adapters/transcription.js.map +0 -1
- package/dist/esm/adapters/tts.d.ts +0 -70
- package/dist/esm/adapters/tts.js +0 -142
- package/dist/esm/adapters/tts.js.map +0 -1
- package/dist/esm/adapters/video.d.ts +0 -128
- package/dist/esm/adapters/video.js +0 -237
- package/dist/esm/adapters/video.js.map +0 -1
- package/dist/esm/audio/transcription-provider-options.d.ts +0 -41
- package/dist/esm/audio/tts-provider-options.d.ts +0 -42
- package/dist/esm/image/image-provider-options.d.ts +0 -133
- package/dist/esm/image/image-provider-options.js +0 -76
- package/dist/esm/image/image-provider-options.js.map +0 -1
- package/dist/esm/index.d.ts +0 -16
- package/dist/esm/index.js +0 -40
- package/dist/esm/index.js.map +0 -1
- package/dist/esm/message-types.d.ts +0 -64
- package/dist/esm/model-meta.d.ts +0 -99
- package/dist/esm/model-meta.js +0 -58
- package/dist/esm/model-meta.js.map +0 -1
- package/dist/esm/realtime/adapter.d.ts +0 -21
- package/dist/esm/realtime/adapter.js +0 -823
- package/dist/esm/realtime/adapter.js.map +0 -1
- package/dist/esm/realtime/index.d.ts +0 -4
- package/dist/esm/realtime/token.d.ts +0 -22
- package/dist/esm/realtime/token.js +0 -75
- package/dist/esm/realtime/token.js.map +0 -1
- package/dist/esm/realtime/types.d.ts +0 -95
- package/dist/esm/text/text-provider-options.d.ts +0 -54
- package/dist/esm/tools/index.d.ts +0 -45
- package/dist/esm/tools/index.js +0 -113
- package/dist/esm/tools/index.js.map +0 -1
- package/dist/esm/utils/audio.d.ts +0 -23
- package/dist/esm/utils/audio.js +0 -172
- package/dist/esm/utils/audio.js.map +0 -1
- package/dist/esm/utils/client.d.ts +0 -14
- package/dist/esm/utils/client.js +0 -21
- package/dist/esm/utils/client.js.map +0 -1
- package/dist/esm/utils/index.d.ts +0 -4
- package/dist/esm/utils/schema-converter.d.ts +0 -2
- package/dist/esm/video/video-provider-options.d.ts +0 -135
- package/dist/esm/video/video-provider-options.js +0 -67
- package/dist/esm/video/video-provider-options.js.map +0 -1
|
@@ -1,122 +0,0 @@
|
|
|
1
|
-
import { BaseTranscriptionAdapter } from "@tanstack/ai/adapters";
|
|
2
|
-
import { generateId } from "@tanstack/ai-utils";
|
|
3
|
-
import { getGrokApiKeyFromEnv } from "../utils/client.js";
|
|
4
|
-
import "@tanstack/openai-base";
|
|
5
|
-
import { toAudioFile } from "../utils/audio.js";
|
|
6
|
-
const DEFAULT_GROK_BASE_URL = "https://api.x.ai/v1";
|
|
7
|
-
class GrokTranscriptionAdapter extends BaseTranscriptionAdapter {
|
|
8
|
-
name = "grok";
|
|
9
|
-
apiKey;
|
|
10
|
-
baseURL;
|
|
11
|
-
defaultHeaders;
|
|
12
|
-
constructor(config, model) {
|
|
13
|
-
super(model, config);
|
|
14
|
-
this.apiKey = config.apiKey;
|
|
15
|
-
this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, "");
|
|
16
|
-
this.defaultHeaders = config.defaultHeaders ?? {};
|
|
17
|
-
}
|
|
18
|
-
async transcribe(options) {
|
|
19
|
-
const { logger } = options;
|
|
20
|
-
const { model, audio, language, modelOptions } = options;
|
|
21
|
-
logger.request(
|
|
22
|
-
`activity=generateTranscription provider=grok model=${model}`,
|
|
23
|
-
{ provider: "grok", model }
|
|
24
|
-
);
|
|
25
|
-
const file = toAudioFile(audio, modelOptions?.audio_format);
|
|
26
|
-
const form = buildTranscriptionFormData({ file, language, modelOptions });
|
|
27
|
-
try {
|
|
28
|
-
const response = await fetch(`${this.baseURL}/stt`, {
|
|
29
|
-
method: "POST",
|
|
30
|
-
headers: {
|
|
31
|
-
// `defaultHeaders` first so Authorization always wins.
|
|
32
|
-
...this.defaultHeaders,
|
|
33
|
-
Authorization: `Bearer ${this.apiKey}`
|
|
34
|
-
},
|
|
35
|
-
body: form
|
|
36
|
-
});
|
|
37
|
-
if (!response.ok) {
|
|
38
|
-
const errorText = await response.text();
|
|
39
|
-
throw new Error(
|
|
40
|
-
`Grok transcription request failed: ${response.status} ${errorText}`
|
|
41
|
-
);
|
|
42
|
-
}
|
|
43
|
-
const data = await response.json();
|
|
44
|
-
const words = data.words?.map(
|
|
45
|
-
(w) => {
|
|
46
|
-
const tw = {
|
|
47
|
-
word: w.text,
|
|
48
|
-
start: w.start,
|
|
49
|
-
end: w.end
|
|
50
|
-
};
|
|
51
|
-
if (w.confidence !== void 0) tw.confidence = w.confidence;
|
|
52
|
-
if (w.speaker !== void 0) tw.speaker = w.speaker;
|
|
53
|
-
return tw;
|
|
54
|
-
}
|
|
55
|
-
);
|
|
56
|
-
const resolvedLanguage = data.language ?? language;
|
|
57
|
-
const usage = data.duration !== void 0 && data.duration > 0 ? {
|
|
58
|
-
promptTokens: 0,
|
|
59
|
-
completionTokens: 0,
|
|
60
|
-
totalTokens: 0,
|
|
61
|
-
durationSeconds: data.duration
|
|
62
|
-
} : void 0;
|
|
63
|
-
return {
|
|
64
|
-
id: generateId(this.name),
|
|
65
|
-
model,
|
|
66
|
-
text: data.text,
|
|
67
|
-
...resolvedLanguage !== void 0 && { language: resolvedLanguage },
|
|
68
|
-
duration: data.duration,
|
|
69
|
-
...words !== void 0 && { words },
|
|
70
|
-
...usage !== void 0 && { usage }
|
|
71
|
-
};
|
|
72
|
-
} catch (error) {
|
|
73
|
-
logger.errors("grok.transcribe fatal", {
|
|
74
|
-
error,
|
|
75
|
-
source: "grok.transcribe"
|
|
76
|
-
});
|
|
77
|
-
throw error;
|
|
78
|
-
}
|
|
79
|
-
}
|
|
80
|
-
}
|
|
81
|
-
function buildTranscriptionFormData(options) {
|
|
82
|
-
const { file, language, modelOptions } = options;
|
|
83
|
-
const form = new FormData();
|
|
84
|
-
form.set("file", file);
|
|
85
|
-
if (language) form.set("language", language);
|
|
86
|
-
if (modelOptions?.audio_format !== void 0) {
|
|
87
|
-
form.set("audio_format", modelOptions.audio_format);
|
|
88
|
-
}
|
|
89
|
-
if (modelOptions?.sample_rate !== void 0) {
|
|
90
|
-
form.set("sample_rate", String(modelOptions.sample_rate));
|
|
91
|
-
}
|
|
92
|
-
if (modelOptions?.inverse_text_normalization !== void 0) {
|
|
93
|
-
form.set(
|
|
94
|
-
"format",
|
|
95
|
-
modelOptions.inverse_text_normalization ? "true" : "false"
|
|
96
|
-
);
|
|
97
|
-
}
|
|
98
|
-
if (modelOptions?.multichannel !== void 0) {
|
|
99
|
-
form.set("multichannel", modelOptions.multichannel ? "true" : "false");
|
|
100
|
-
}
|
|
101
|
-
if (modelOptions?.channels !== void 0) {
|
|
102
|
-
form.set("channels", String(modelOptions.channels));
|
|
103
|
-
}
|
|
104
|
-
if (modelOptions?.diarize !== void 0) {
|
|
105
|
-
form.set("diarize", modelOptions.diarize ? "true" : "false");
|
|
106
|
-
}
|
|
107
|
-
return form;
|
|
108
|
-
}
|
|
109
|
-
function createGrokTranscription(model, apiKey, config) {
|
|
110
|
-
return new GrokTranscriptionAdapter({ apiKey, ...config }, model);
|
|
111
|
-
}
|
|
112
|
-
function grokTranscription(model, config) {
|
|
113
|
-
const apiKey = getGrokApiKeyFromEnv();
|
|
114
|
-
return createGrokTranscription(model, apiKey, config);
|
|
115
|
-
}
|
|
116
|
-
export {
|
|
117
|
-
GrokTranscriptionAdapter,
|
|
118
|
-
buildTranscriptionFormData,
|
|
119
|
-
createGrokTranscription,
|
|
120
|
-
grokTranscription
|
|
121
|
-
};
|
|
122
|
-
//# sourceMappingURL=transcription.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"transcription.js","sources":["../../../src/adapters/transcription.ts"],"sourcesContent":["import { BaseTranscriptionAdapter } from '@tanstack/ai/adapters'\nimport { generateId, getGrokApiKeyFromEnv, toAudioFile } from '../utils'\nimport type {\n TokenUsage,\n TranscriptionOptions,\n TranscriptionResult,\n TranscriptionWord,\n} from '@tanstack/ai'\nimport type { GrokTranscriptionModel } from '../model-meta'\nimport type { GrokTranscriptionProviderOptions } from '../audio/transcription-provider-options'\n\n/**\n * Grok-specific extension of `TranscriptionWord` that surfaces the extra\n * fields xAI returns when diarization / confidence are enabled. The base\n * cross-provider `TranscriptionWord` contract doesn't include these, so\n * callers who know they're using Grok can narrow with:\n *\n * ```ts\n * const words = result.words as Array<GrokTranscriptionWord> | undefined\n * ```\n */\nexport interface GrokTranscriptionWord extends TranscriptionWord {\n /** Model confidence for the word, when xAI returns one. */\n confidence?: number\n /** Speaker index, populated when `modelOptions.diarize === true`. */\n speaker?: number\n}\n\nconst DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'\n\n/**\n * Configuration for the Grok transcription adapter.\n *\n * Uses direct `fetch` rather than the OpenAI SDK because xAI's `/v1/stt`\n * endpoint is not OpenAI-compatible.\n */\nexport interface GrokTranscriptionConfig {\n apiKey: string\n baseURL?: string\n /** Additional headers to merge into every request (e.g., test IDs). */\n defaultHeaders?: Record<string, string>\n}\n\n/**\n * xAI STT response shape from `POST /v1/stt`.\n * Grok returns word-level timestamps only; no segment array.\n */\ninterface GrokSTTWord {\n text: string\n start: number\n end: number\n confidence?: number\n speaker?: number\n}\n\ninterface GrokSTTResponse {\n text: string\n language?: string\n duration?: number\n words?: Array<GrokSTTWord>\n channels?: Array<unknown>\n}\n\n/**\n * Grok Speech-to-Text Adapter.\n *\n * Talks to `POST {baseURL}/stt` per\n * https://docs.x.ai/developers/rest-api-reference/inference/voice\n */\nexport class GrokTranscriptionAdapter<\n TModel extends GrokTranscriptionModel,\n> extends BaseTranscriptionAdapter<TModel, GrokTranscriptionProviderOptions> {\n readonly name = 'grok' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n\n constructor(config: GrokTranscriptionConfig, model: TModel) {\n super(model, config)\n this.apiKey = config.apiKey\n this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\\/+$/, '')\n this.defaultHeaders = config.defaultHeaders ?? {}\n }\n\n async transcribe(\n options: TranscriptionOptions<GrokTranscriptionProviderOptions>,\n ): Promise<TranscriptionResult> {\n const { logger } = options\n const { model, audio, language, modelOptions } = options\n\n logger.request(\n `activity=generateTranscription provider=grok model=${model}`,\n { provider: 'grok', model },\n )\n\n const file = toAudioFile(audio, modelOptions?.audio_format)\n const form = buildTranscriptionFormData({ file, language, modelOptions })\n\n try {\n const response = await fetch(`${this.baseURL}/stt`, {\n method: 'POST',\n headers: {\n // `defaultHeaders` first so Authorization always wins.\n ...this.defaultHeaders,\n Authorization: `Bearer ${this.apiKey}`,\n },\n body: form,\n })\n\n if (!response.ok) {\n const errorText = await response.text()\n throw new Error(\n `Grok transcription request failed: ${response.status} ${errorText}`,\n )\n }\n\n const data = (await response.json()) as GrokSTTResponse\n\n const words: Array<TranscriptionWord> | undefined = data.words?.map(\n (w) => {\n // Construct a GrokTranscriptionWord so that `confidence` and\n // `speaker` (when xAI returns them under `diarize` / confidence\n // mode) are preserved on the result. The returned array is typed\n // as `Array<TranscriptionWord>` per the cross-provider contract;\n // callers who want the extras narrow via `as Array<GrokTranscriptionWord>`.\n const tw: GrokTranscriptionWord = {\n word: w.text,\n start: w.start,\n end: w.end,\n }\n if (w.confidence !== undefined) tw.confidence = w.confidence\n if (w.speaker !== undefined) tw.speaker = w.speaker\n return tw\n },\n )\n\n const resolvedLanguage = data.language ?? language\n // xAI's /v1/stt response carries no token counts — STT is duration-billed —\n // so surface the audio duration as `durationSeconds`, mirroring the\n // whisper-1 path in the OpenAI transcription adapter.\n const usage: TokenUsage | undefined =\n data.duration !== undefined && data.duration > 0\n ? {\n promptTokens: 0,\n completionTokens: 0,\n totalTokens: 0,\n durationSeconds: data.duration,\n }\n : undefined\n return {\n id: generateId(this.name),\n model,\n text: data.text,\n ...(resolvedLanguage !== undefined && { language: resolvedLanguage }),\n duration: data.duration,\n ...(words !== undefined && { words }),\n ...(usage !== undefined && { usage }),\n }\n } catch (error) {\n logger.errors('grok.transcribe fatal', {\n error,\n source: 'grok.transcribe',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the multipart/form-data body for `POST /v1/stt`, coercing SDK-level\n * model options into xAI's wire format (booleans as `'true'`/`'false'`\n * strings, numeric fields stringified, etc.).\n *\n * Wire-field mapping:\n * - `modelOptions.inverse_text_normalization` → `format` (xAI's chosen\n * wire-field name for the ITN boolean; the SDK surfaces it under the\n * clearer `inverse_text_normalization` key).\n * - `modelOptions.audio_format`, `sample_rate`, `multichannel`, `channels`,\n * `diarize` map to same-named form fields.\n */\nexport function buildTranscriptionFormData(options: {\n file: File\n language: string | undefined\n modelOptions: GrokTranscriptionProviderOptions | undefined\n}): FormData {\n const { file, language, modelOptions } = options\n const form = new FormData()\n form.set('file', file)\n if (language) form.set('language', language)\n if (modelOptions?.audio_format !== undefined) {\n form.set('audio_format', modelOptions.audio_format)\n }\n if (modelOptions?.sample_rate !== undefined) {\n form.set('sample_rate', String(modelOptions.sample_rate))\n }\n if (modelOptions?.inverse_text_normalization !== undefined) {\n form.set(\n 'format',\n modelOptions.inverse_text_normalization ? 'true' : 'false',\n )\n }\n if (modelOptions?.multichannel !== undefined) {\n form.set('multichannel', modelOptions.multichannel ? 'true' : 'false')\n }\n if (modelOptions?.channels !== undefined) {\n form.set('channels', String(modelOptions.channels))\n }\n if (modelOptions?.diarize !== undefined) {\n form.set('diarize', modelOptions.diarize ? 'true' : 'false')\n }\n return form\n}\n\n/**\n * Creates a Grok transcription adapter with an explicit API key.\n *\n * @example\n * ```typescript\n * const adapter = createGrokTranscription('grok-stt', 'xai-...')\n * const result = await generateTranscription({\n * adapter,\n * audio: audioFile,\n * language: 'en',\n * })\n * ```\n */\nexport function createGrokTranscription<TModel extends GrokTranscriptionModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GrokTranscriptionConfig, 'apiKey'>,\n): GrokTranscriptionAdapter<TModel> {\n return new GrokTranscriptionAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Grok transcription adapter, reading the API key from\n * `XAI_API_KEY` in the environment.\n *\n * @throws Error if `XAI_API_KEY` is not set.\n */\nexport function grokTranscription<TModel extends GrokTranscriptionModel>(\n model: TModel,\n config?: Omit<GrokTranscriptionConfig, 'apiKey'>,\n): GrokTranscriptionAdapter<TModel> {\n const apiKey = getGrokApiKeyFromEnv()\n return createGrokTranscription(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;;;AA4BA,MAAM,wBAAwB;AAyCvB,MAAM,iCAEH,yBAAmE;AAAA,EAClE,OAAO;AAAA,EAEC;AAAA,EACA;AAAA,EACA;AAAA,EAEjB,YAAY,QAAiC,OAAe;AAC1D,UAAM,OAAO,MAAM;AACnB,SAAK,SAAS,OAAO;AACrB,SAAK,WAAW,OAAO,WAAW,uBAAuB,QAAQ,QAAQ,EAAE;AAC3E,SAAK,iBAAiB,OAAO,kBAAkB,CAAA;AAAA,EACjD;AAAA,EAEA,MAAM,WACJ,SAC8B;AAC9B,UAAM,EAAE,WAAW;AACnB,UAAM,EAAE,OAAO,OAAO,UAAU,iBAAiB;AAEjD,WAAO;AAAA,MACL,sDAAsD,KAAK;AAAA,MAC3D,EAAE,UAAU,QAAQ,MAAA;AAAA,IAAM;AAG5B,UAAM,OAAO,YAAY,OAAO,cAAc,YAAY;AAC1D,UAAM,OAAO,2BAA2B,EAAE,MAAM,UAAU,cAAc;AAExE,QAAI;AACF,YAAM,WAAW,MAAM,MAAM,GAAG,KAAK,OAAO,QAAQ;AAAA,QAClD,QAAQ;AAAA,QACR,SAAS;AAAA;AAAA,UAEP,GAAG,KAAK;AAAA,UACR,eAAe,UAAU,KAAK,MAAM;AAAA,QAAA;AAAA,QAEtC,MAAM;AAAA,MAAA,CACP;AAED,UAAI,CAAC,SAAS,IAAI;AAChB,cAAM,YAAY,MAAM,SAAS,KAAA;AACjC,cAAM,IAAI;AAAA,UACR,sCAAsC,SAAS,MAAM,IAAI,SAAS;AAAA,QAAA;AAAA,MAEtE;AAEA,YAAM,OAAQ,MAAM,SAAS,KAAA;AAE7B,YAAM,QAA8C,KAAK,OAAO;AAAA,QAC9D,CAAC,MAAM;AAML,gBAAM,KAA4B;AAAA,YAChC,MAAM,EAAE;AAAA,YACR,OAAO,EAAE;AAAA,YACT,KAAK,EAAE;AAAA,UAAA;AAET,cAAI,EAAE,eAAe,OAAW,IAAG,aAAa,EAAE;AAClD,cAAI,EAAE,YAAY,OAAW,IAAG,UAAU,EAAE;AAC5C,iBAAO;AAAA,QACT;AAAA,MAAA;AAGF,YAAM,mBAAmB,KAAK,YAAY;AAI1C,YAAM,QACJ,KAAK,aAAa,UAAa,KAAK,WAAW,IAC3C;AAAA,QACE,cAAc;AAAA,QACd,kBAAkB;AAAA,QAClB,aAAa;AAAA,QACb,iBAAiB,KAAK;AAAA,MAAA,IAExB;AACN,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA,MAAM,KAAK;AAAA,QACX,GAAI,qBAAqB,UAAa,EAAE,UAAU,iBAAA;AAAA,QAClD,UAAU,KAAK;AAAA,QACf,GAAI,UAAU,UAAa,EAAE,MAAA;AAAA,QAC7B,GAAI,UAAU,UAAa,EAAE,MAAA;AAAA,MAAM;AAAA,IAEvC,SAAS,OAAO;AACd,aAAO,OAAO,yBAAyB;AAAA,QACrC;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AAcO,SAAS,2BAA2B,SAI9B;AACX,QAAM,EAAE,MAAM,UAAU,aAAA,IAAiB;AACzC,QAAM,OAAO,IAAI,SAAA;AACjB,OAAK,IAAI,QAAQ,IAAI;AACrB,MAAI,SAAU,MAAK,IAAI,YAAY,QAAQ;AAC3C,MAAI,cAAc,iBAAiB,QAAW;AAC5C,SAAK,IAAI,gBAAgB,aAAa,YAAY;AAAA,EACpD;AACA,MAAI,cAAc,gBAAgB,QAAW;AAC3C,SAAK,IAAI,eAAe,OAAO,aAAa,WAAW,CAAC;AAAA,EAC1D;AACA,MAAI,cAAc,+BAA+B,QAAW;AAC1D,SAAK;AAAA,MACH;AAAA,MACA,aAAa,6BAA6B,SAAS;AAAA,IAAA;AAAA,EAEvD;AACA,MAAI,cAAc,iBAAiB,QAAW;AAC5C,SAAK,IAAI,gBAAgB,aAAa,eAAe,SAAS,OAAO;AAAA,EACvE;AACA,MAAI,cAAc,aAAa,QAAW;AACxC,SAAK,IAAI,YAAY,OAAO,aAAa,QAAQ,CAAC;AAAA,EACpD;AACA,MAAI,cAAc,YAAY,QAAW;AACvC,SAAK,IAAI,WAAW,aAAa,UAAU,SAAS,OAAO;AAAA,EAC7D;AACA,SAAO;AACT;AAeO,SAAS,wBACd,OACA,QACA,QACkC;AAClC,SAAO,IAAI,yBAAyB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAClE;AAQO,SAAS,kBACd,OACA,QACkC;AAClC,QAAM,SAAS,qBAAA;AACf,SAAO,wBAAwB,OAAO,QAAQ,MAAM;AACtD;"}
|
|
@@ -1,70 +0,0 @@
|
|
|
1
|
-
import { BaseTTSAdapter } from '@tanstack/ai/adapters';
|
|
2
|
-
import { TTSOptions, TTSResult } from '@tanstack/ai';
|
|
3
|
-
import { GrokTTSModel } from '../model-meta.js';
|
|
4
|
-
import { GrokTTSCodec, GrokTTSProviderOptions } from '../audio/tts-provider-options.js';
|
|
5
|
-
/**
|
|
6
|
-
* Configuration for the Grok TTS adapter.
|
|
7
|
-
*
|
|
8
|
-
* Unlike chat/image/summarize adapters, TTS does not use the OpenAI SDK
|
|
9
|
-
* because xAI's `/v1/tts` endpoint is not OpenAI-compatible. This config
|
|
10
|
-
* is a minimal subset suitable for direct `fetch` calls.
|
|
11
|
-
*/
|
|
12
|
-
export interface GrokSpeechConfig {
|
|
13
|
-
apiKey: string;
|
|
14
|
-
baseURL?: string;
|
|
15
|
-
/** Additional headers to merge into every request (e.g., test IDs). */
|
|
16
|
-
defaultHeaders?: Record<string, string>;
|
|
17
|
-
}
|
|
18
|
-
/**
|
|
19
|
-
* Grok Text-to-Speech Adapter.
|
|
20
|
-
*
|
|
21
|
-
* Talks to `POST {baseURL}/tts` per
|
|
22
|
-
* https://docs.x.ai/developers/model-capabilities/audio/text-to-speech
|
|
23
|
-
*/
|
|
24
|
-
export declare class GrokSpeechAdapter<TModel extends GrokTTSModel> extends BaseTTSAdapter<TModel, GrokTTSProviderOptions> {
|
|
25
|
-
readonly name: "grok";
|
|
26
|
-
private readonly apiKey;
|
|
27
|
-
private readonly baseURL;
|
|
28
|
-
private readonly defaultHeaders;
|
|
29
|
-
constructor(config: GrokSpeechConfig, model: TModel);
|
|
30
|
-
generateSpeech(options: TTSOptions<GrokTTSProviderOptions>): Promise<TTSResult>;
|
|
31
|
-
}
|
|
32
|
-
/**
|
|
33
|
-
* Build the JSON body for `POST /v1/tts`, resolving codec / sample-rate / voice
|
|
34
|
-
* defaults in one place.
|
|
35
|
-
*
|
|
36
|
-
* Returns the request `body`, the resolved `codec`, and the `sampleRateForContentType`
|
|
37
|
-
* used by the caller to label the response via `getContentType`.
|
|
38
|
-
*/
|
|
39
|
-
export declare function buildTTSRequestBody(options: {
|
|
40
|
-
text: string;
|
|
41
|
-
voice: string | undefined;
|
|
42
|
-
format: TTSOptions['format'] | undefined;
|
|
43
|
-
modelOptions: GrokTTSProviderOptions | undefined;
|
|
44
|
-
}): {
|
|
45
|
-
body: Record<string, unknown>;
|
|
46
|
-
codec: GrokTTSCodec;
|
|
47
|
-
sampleRateForContentType: number;
|
|
48
|
-
};
|
|
49
|
-
export declare function getContentType(codec: GrokTTSCodec, sampleRate: number): string;
|
|
50
|
-
/**
|
|
51
|
-
* Creates a Grok speech (TTS) adapter with an explicit API key.
|
|
52
|
-
*
|
|
53
|
-
* @example
|
|
54
|
-
* ```typescript
|
|
55
|
-
* const adapter = createGrokSpeech('grok-tts', 'xai-...')
|
|
56
|
-
* const result = await generateSpeech({
|
|
57
|
-
* adapter,
|
|
58
|
-
* text: 'Hello from Grok',
|
|
59
|
-
* voice: 'eve',
|
|
60
|
-
* })
|
|
61
|
-
* ```
|
|
62
|
-
*/
|
|
63
|
-
export declare function createGrokSpeech<TModel extends GrokTTSModel>(model: TModel, apiKey: string, config?: Omit<GrokSpeechConfig, 'apiKey'>): GrokSpeechAdapter<TModel>;
|
|
64
|
-
/**
|
|
65
|
-
* Creates a Grok speech (TTS) adapter, reading the API key from
|
|
66
|
-
* `XAI_API_KEY` in the environment.
|
|
67
|
-
*
|
|
68
|
-
* @throws Error if `XAI_API_KEY` is not set.
|
|
69
|
-
*/
|
|
70
|
-
export declare function grokSpeech<TModel extends GrokTTSModel>(model: TModel, config?: Omit<GrokSpeechConfig, 'apiKey'>): GrokSpeechAdapter<TModel>;
|
package/dist/esm/adapters/tts.js
DELETED
|
@@ -1,142 +0,0 @@
|
|
|
1
|
-
import { BaseTTSAdapter } from "@tanstack/ai/adapters";
|
|
2
|
-
import { generateId } from "@tanstack/ai-utils";
|
|
3
|
-
import { getGrokApiKeyFromEnv } from "../utils/client.js";
|
|
4
|
-
import "@tanstack/openai-base";
|
|
5
|
-
import { arrayBufferToBase64 } from "../utils/audio.js";
|
|
6
|
-
const DEFAULT_GROK_BASE_URL = "https://api.x.ai/v1";
|
|
7
|
-
class GrokSpeechAdapter extends BaseTTSAdapter {
|
|
8
|
-
name = "grok";
|
|
9
|
-
apiKey;
|
|
10
|
-
baseURL;
|
|
11
|
-
defaultHeaders;
|
|
12
|
-
constructor(config, model) {
|
|
13
|
-
super(model, config);
|
|
14
|
-
this.apiKey = config.apiKey;
|
|
15
|
-
this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\/+$/, "");
|
|
16
|
-
this.defaultHeaders = config.defaultHeaders ?? {};
|
|
17
|
-
}
|
|
18
|
-
async generateSpeech(options) {
|
|
19
|
-
const { logger } = options;
|
|
20
|
-
const { model, text, voice, format, modelOptions } = options;
|
|
21
|
-
logger.request(`activity=generateSpeech provider=grok model=${model}`, {
|
|
22
|
-
provider: "grok",
|
|
23
|
-
model
|
|
24
|
-
});
|
|
25
|
-
const { body, codec, sampleRateForContentType } = buildTTSRequestBody({
|
|
26
|
-
text,
|
|
27
|
-
voice,
|
|
28
|
-
format,
|
|
29
|
-
modelOptions
|
|
30
|
-
});
|
|
31
|
-
try {
|
|
32
|
-
const response = await fetch(`${this.baseURL}/tts`, {
|
|
33
|
-
method: "POST",
|
|
34
|
-
headers: {
|
|
35
|
-
// `defaultHeaders` first so the adapter's Authorization / Content-Type
|
|
36
|
-
// always win — otherwise a caller-supplied `Authorization` header
|
|
37
|
-
// could silently clobber the bearer token.
|
|
38
|
-
...this.defaultHeaders,
|
|
39
|
-
Authorization: `Bearer ${this.apiKey}`,
|
|
40
|
-
"Content-Type": "application/json"
|
|
41
|
-
},
|
|
42
|
-
body: JSON.stringify(body)
|
|
43
|
-
});
|
|
44
|
-
if (!response.ok) {
|
|
45
|
-
const errorText = await response.text();
|
|
46
|
-
throw new Error(
|
|
47
|
-
`Grok TTS request failed: ${response.status} ${errorText}`
|
|
48
|
-
);
|
|
49
|
-
}
|
|
50
|
-
const arrayBuffer = await response.arrayBuffer();
|
|
51
|
-
const audio = arrayBufferToBase64(arrayBuffer);
|
|
52
|
-
return {
|
|
53
|
-
id: generateId(this.name),
|
|
54
|
-
model,
|
|
55
|
-
audio,
|
|
56
|
-
format: codec,
|
|
57
|
-
contentType: getContentType(codec, sampleRateForContentType)
|
|
58
|
-
};
|
|
59
|
-
} catch (error) {
|
|
60
|
-
logger.errors("grok.generateSpeech fatal", {
|
|
61
|
-
error,
|
|
62
|
-
source: "grok.generateSpeech"
|
|
63
|
-
});
|
|
64
|
-
throw error;
|
|
65
|
-
}
|
|
66
|
-
}
|
|
67
|
-
}
|
|
68
|
-
function buildTTSRequestBody(options) {
|
|
69
|
-
const { text, voice, format, modelOptions } = options;
|
|
70
|
-
const codec = pickCodec(modelOptions?.codec, format);
|
|
71
|
-
const callerSampleRate = modelOptions?.sample_rate;
|
|
72
|
-
const pcmDefault = 24e3;
|
|
73
|
-
const needsRateInContentType = codec === "pcm";
|
|
74
|
-
const outputFormat = { codec };
|
|
75
|
-
if (callerSampleRate !== void 0) {
|
|
76
|
-
outputFormat.sample_rate = callerSampleRate;
|
|
77
|
-
} else if (needsRateInContentType) {
|
|
78
|
-
outputFormat.sample_rate = pcmDefault;
|
|
79
|
-
}
|
|
80
|
-
if (codec === "mp3" && modelOptions?.bit_rate !== void 0) {
|
|
81
|
-
outputFormat.bit_rate = modelOptions.bit_rate;
|
|
82
|
-
}
|
|
83
|
-
const sampleRateForContentType = callerSampleRate ?? pcmDefault;
|
|
84
|
-
const body = {
|
|
85
|
-
text,
|
|
86
|
-
voice_id: voice ?? "eve",
|
|
87
|
-
language: modelOptions?.language ?? "en",
|
|
88
|
-
output_format: outputFormat
|
|
89
|
-
};
|
|
90
|
-
if (modelOptions?.optimize_streaming_latency !== void 0) {
|
|
91
|
-
body.optimize_streaming_latency = modelOptions.optimize_streaming_latency;
|
|
92
|
-
}
|
|
93
|
-
if (modelOptions?.text_normalization !== void 0) {
|
|
94
|
-
body.text_normalization = modelOptions.text_normalization;
|
|
95
|
-
}
|
|
96
|
-
return { body, codec, sampleRateForContentType };
|
|
97
|
-
}
|
|
98
|
-
function pickCodec(codecOverride, format) {
|
|
99
|
-
if (codecOverride) return codecOverride;
|
|
100
|
-
if (!format) return "mp3";
|
|
101
|
-
switch (format) {
|
|
102
|
-
case "mp3":
|
|
103
|
-
case "wav":
|
|
104
|
-
case "pcm":
|
|
105
|
-
return format;
|
|
106
|
-
case "flac":
|
|
107
|
-
case "opus":
|
|
108
|
-
case "aac":
|
|
109
|
-
return "mp3";
|
|
110
|
-
default:
|
|
111
|
-
return "mp3";
|
|
112
|
-
}
|
|
113
|
-
}
|
|
114
|
-
function getContentType(codec, sampleRate) {
|
|
115
|
-
switch (codec) {
|
|
116
|
-
case "mp3":
|
|
117
|
-
return "audio/mpeg";
|
|
118
|
-
case "wav":
|
|
119
|
-
return "audio/wav";
|
|
120
|
-
case "pcm":
|
|
121
|
-
return `audio/L16;rate=${sampleRate}`;
|
|
122
|
-
case "mulaw":
|
|
123
|
-
return sampleRate === 8e3 ? "audio/basic" : `audio/PCMU;rate=${sampleRate}`;
|
|
124
|
-
case "alaw":
|
|
125
|
-
return sampleRate === 8e3 ? "audio/x-alaw-basic" : `audio/PCMA;rate=${sampleRate}`;
|
|
126
|
-
}
|
|
127
|
-
}
|
|
128
|
-
function createGrokSpeech(model, apiKey, config) {
|
|
129
|
-
return new GrokSpeechAdapter({ apiKey, ...config }, model);
|
|
130
|
-
}
|
|
131
|
-
function grokSpeech(model, config) {
|
|
132
|
-
const apiKey = getGrokApiKeyFromEnv();
|
|
133
|
-
return createGrokSpeech(model, apiKey, config);
|
|
134
|
-
}
|
|
135
|
-
export {
|
|
136
|
-
GrokSpeechAdapter,
|
|
137
|
-
buildTTSRequestBody,
|
|
138
|
-
createGrokSpeech,
|
|
139
|
-
getContentType,
|
|
140
|
-
grokSpeech
|
|
141
|
-
};
|
|
142
|
-
//# sourceMappingURL=tts.js.map
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"file":"tts.js","sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport { arrayBufferToBase64, generateId, getGrokApiKeyFromEnv } from '../utils'\nimport type { TTSOptions, TTSResult } from '@tanstack/ai'\nimport type { GrokTTSModel } from '../model-meta'\nimport type {\n GrokTTSCodec,\n GrokTTSProviderOptions,\n} from '../audio/tts-provider-options'\n\nconst DEFAULT_GROK_BASE_URL = 'https://api.x.ai/v1'\n\n/**\n * Configuration for the Grok TTS adapter.\n *\n * Unlike chat/image/summarize adapters, TTS does not use the OpenAI SDK\n * because xAI's `/v1/tts` endpoint is not OpenAI-compatible. This config\n * is a minimal subset suitable for direct `fetch` calls.\n */\nexport interface GrokSpeechConfig {\n apiKey: string\n baseURL?: string\n /** Additional headers to merge into every request (e.g., test IDs). */\n defaultHeaders?: Record<string, string>\n}\n\n/**\n * Grok Text-to-Speech Adapter.\n *\n * Talks to `POST {baseURL}/tts` per\n * https://docs.x.ai/developers/model-capabilities/audio/text-to-speech\n */\nexport class GrokSpeechAdapter<\n TModel extends GrokTTSModel,\n> extends BaseTTSAdapter<TModel, GrokTTSProviderOptions> {\n readonly name = 'grok' as const\n\n private readonly apiKey: string\n private readonly baseURL: string\n private readonly defaultHeaders: Record<string, string>\n\n constructor(config: GrokSpeechConfig, model: TModel) {\n super(model, config)\n this.apiKey = config.apiKey\n this.baseURL = (config.baseURL ?? DEFAULT_GROK_BASE_URL).replace(/\\/+$/, '')\n this.defaultHeaders = config.defaultHeaders ?? {}\n }\n\n async generateSpeech(\n options: TTSOptions<GrokTTSProviderOptions>,\n ): Promise<TTSResult> {\n const { logger } = options\n const { model, text, voice, format, modelOptions } = options\n\n logger.request(`activity=generateSpeech provider=grok model=${model}`, {\n provider: 'grok',\n model,\n })\n\n const { body, codec, sampleRateForContentType } = buildTTSRequestBody({\n text,\n voice,\n format,\n modelOptions,\n })\n\n try {\n const response = await fetch(`${this.baseURL}/tts`, {\n method: 'POST',\n headers: {\n // `defaultHeaders` first so the adapter's Authorization / Content-Type\n // always win — otherwise a caller-supplied `Authorization` header\n // could silently clobber the bearer token.\n ...this.defaultHeaders,\n Authorization: `Bearer ${this.apiKey}`,\n 'Content-Type': 'application/json',\n },\n body: JSON.stringify(body),\n })\n\n if (!response.ok) {\n const errorText = await response.text()\n throw new Error(\n `Grok TTS request failed: ${response.status} ${errorText}`,\n )\n }\n\n const arrayBuffer = await response.arrayBuffer()\n const audio = arrayBufferToBase64(arrayBuffer)\n\n return {\n id: generateId(this.name),\n model,\n audio,\n format: codec,\n contentType: getContentType(codec, sampleRateForContentType),\n }\n } catch (error) {\n logger.errors('grok.generateSpeech fatal', {\n error,\n source: 'grok.generateSpeech',\n })\n throw error\n }\n }\n}\n\n/**\n * Build the JSON body for `POST /v1/tts`, resolving codec / sample-rate / voice\n * defaults in one place.\n *\n * Returns the request `body`, the resolved `codec`, and the `sampleRateForContentType`\n * used by the caller to label the response via `getContentType`.\n */\nexport function buildTTSRequestBody(options: {\n text: string\n voice: string | undefined\n format: TTSOptions['format'] | undefined\n modelOptions: GrokTTSProviderOptions | undefined\n}): {\n body: Record<string, unknown>\n codec: GrokTTSCodec\n sampleRateForContentType: number\n} {\n const { text, voice, format, modelOptions } = options\n\n const codec = pickCodec(modelOptions?.codec, format)\n\n // Only forward `sample_rate` when either:\n // - the caller explicitly set `modelOptions.sample_rate`, or\n // - the codec's Content-Type carries the rate (pcm → audio/L16;rate=…).\n // For mp3/wav/opus/aac/flac we leave sample_rate unset so xAI's server\n // default applies.\n const callerSampleRate = modelOptions?.sample_rate\n // Default sample rate documented in GrokTTSProviderOptions is 24000 Hz —\n // used only when we MUST attach a rate to the contentType (pcm) and the\n // caller didn't pick one.\n const pcmDefault = 24000\n const needsRateInContentType = codec === 'pcm'\n\n const outputFormat: Record<string, unknown> = { codec }\n if (callerSampleRate !== undefined) {\n outputFormat.sample_rate = callerSampleRate\n } else if (needsRateInContentType) {\n outputFormat.sample_rate = pcmDefault\n }\n if (codec === 'mp3' && modelOptions?.bit_rate !== undefined) {\n outputFormat.bit_rate = modelOptions.bit_rate\n }\n\n // pcm embeds the rate in `audio/L16;rate=…`; mulaw/alaw embed it in\n // `audio/PCMU;rate=…` / `audio/PCMA;rate=…` when non-default. mp3/wav\n // don't carry a rate parameter so the value is unused for those.\n const sampleRateForContentType = callerSampleRate ?? pcmDefault\n\n const body: Record<string, unknown> = {\n text,\n voice_id: voice ?? 'eve',\n language: modelOptions?.language ?? 'en',\n output_format: outputFormat,\n }\n if (modelOptions?.optimize_streaming_latency !== undefined) {\n body.optimize_streaming_latency = modelOptions.optimize_streaming_latency\n }\n if (modelOptions?.text_normalization !== undefined) {\n body.text_normalization = modelOptions.text_normalization\n }\n\n return { body, codec, sampleRateForContentType }\n}\n\n/**\n * Maps the cross-provider `TTSOptions.format` onto Grok's supported codecs.\n * `opus`, `aac`, and `flac` are not supported by xAI TTS (which only exposes\n * mp3/wav/pcm/mulaw/alaw) — we fall back to mp3. An explicit\n * `modelOptions.codec` always wins.\n */\nfunction pickCodec(\n codecOverride: GrokTTSCodec | undefined,\n format: TTSOptions['format'] | undefined,\n): GrokTTSCodec {\n if (codecOverride) return codecOverride\n if (!format) return 'mp3'\n switch (format) {\n case 'mp3':\n case 'wav':\n case 'pcm':\n return format\n case 'flac':\n case 'opus':\n case 'aac':\n return 'mp3'\n default:\n return 'mp3'\n }\n}\n\nexport function getContentType(\n codec: GrokTTSCodec,\n sampleRate: number,\n): string {\n switch (codec) {\n case 'mp3':\n return 'audio/mpeg'\n case 'wav':\n return 'audio/wav'\n case 'pcm':\n // `audio/L16` requires a `rate` parameter per RFC 3551/3555.\n return `audio/L16;rate=${sampleRate}`\n case 'mulaw':\n // `audio/basic` is 8 kHz mono by RFC 2046 registration. For non-8kHz\n // streams xAI still produces mulaw-encoded bytes at the requested\n // rate, but the registered MIME can't carry that rate — so we use\n // the non-standard but commonly-supported `audio/PCMU;rate=…` (RFC 3551\n // RTP payload name) whenever the caller asked for a rate other than\n // 8000, and keep `audio/basic` for the standard 8kHz case.\n return sampleRate === 8000\n ? 'audio/basic'\n : `audio/PCMU;rate=${sampleRate}`\n case 'alaw':\n return sampleRate === 8000\n ? 'audio/x-alaw-basic'\n : `audio/PCMA;rate=${sampleRate}`\n }\n}\n\n/**\n * Creates a Grok speech (TTS) adapter with an explicit API key.\n *\n * @example\n * ```typescript\n * const adapter = createGrokSpeech('grok-tts', 'xai-...')\n * const result = await generateSpeech({\n * adapter,\n * text: 'Hello from Grok',\n * voice: 'eve',\n * })\n * ```\n */\nexport function createGrokSpeech<TModel extends GrokTTSModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GrokSpeechConfig, 'apiKey'>,\n): GrokSpeechAdapter<TModel> {\n return new GrokSpeechAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Grok speech (TTS) adapter, reading the API key from\n * `XAI_API_KEY` in the environment.\n *\n * @throws Error if `XAI_API_KEY` is not set.\n */\nexport function grokSpeech<TModel extends GrokTTSModel>(\n model: TModel,\n config?: Omit<GrokSpeechConfig, 'apiKey'>,\n): GrokSpeechAdapter<TModel> {\n const apiKey = getGrokApiKeyFromEnv()\n return createGrokSpeech(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;;;AASA,MAAM,wBAAwB;AAsBvB,MAAM,0BAEH,eAA+C;AAAA,EAC9C,OAAO;AAAA,EAEC;AAAA,EACA;AAAA,EACA;AAAA,EAEjB,YAAY,QAA0B,OAAe;AACnD,UAAM,OAAO,MAAM;AACnB,SAAK,SAAS,OAAO;AACrB,SAAK,WAAW,OAAO,WAAW,uBAAuB,QAAQ,QAAQ,EAAE;AAC3E,SAAK,iBAAiB,OAAO,kBAAkB,CAAA;AAAA,EACjD;AAAA,EAEA,MAAM,eACJ,SACoB;AACpB,UAAM,EAAE,WAAW;AACnB,UAAM,EAAE,OAAO,MAAM,OAAO,QAAQ,iBAAiB;AAErD,WAAO,QAAQ,+CAA+C,KAAK,IAAI;AAAA,MACrE,UAAU;AAAA,MACV;AAAA,IAAA,CACD;AAED,UAAM,EAAE,MAAM,OAAO,yBAAA,IAA6B,oBAAoB;AAAA,MACpE;AAAA,MACA;AAAA,MACA;AAAA,MACA;AAAA,IAAA,CACD;AAED,QAAI;AACF,YAAM,WAAW,MAAM,MAAM,GAAG,KAAK,OAAO,QAAQ;AAAA,QAClD,QAAQ;AAAA,QACR,SAAS;AAAA;AAAA;AAAA;AAAA,UAIP,GAAG,KAAK;AAAA,UACR,eAAe,UAAU,KAAK,MAAM;AAAA,UACpC,gBAAgB;AAAA,QAAA;AAAA,QAElB,MAAM,KAAK,UAAU,IAAI;AAAA,MAAA,CAC1B;AAED,UAAI,CAAC,SAAS,IAAI;AAChB,cAAM,YAAY,MAAM,SAAS,KAAA;AACjC,cAAM,IAAI;AAAA,UACR,4BAA4B,SAAS,MAAM,IAAI,SAAS;AAAA,QAAA;AAAA,MAE5D;AAEA,YAAM,cAAc,MAAM,SAAS,YAAA;AACnC,YAAM,QAAQ,oBAAoB,WAAW;AAE7C,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA;AAAA,QACA,QAAQ;AAAA,QACR,aAAa,eAAe,OAAO,wBAAwB;AAAA,MAAA;AAAA,IAE/D,SAAS,OAAO;AACd,aAAO,OAAO,6BAA6B;AAAA,QACzC;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AASO,SAAS,oBAAoB,SASlC;AACA,QAAM,EAAE,MAAM,OAAO,QAAQ,iBAAiB;AAE9C,QAAM,QAAQ,UAAU,cAAc,OAAO,MAAM;AAOnD,QAAM,mBAAmB,cAAc;AAIvC,QAAM,aAAa;AACnB,QAAM,yBAAyB,UAAU;AAEzC,QAAM,eAAwC,EAAE,MAAA;AAChD,MAAI,qBAAqB,QAAW;AAClC,iBAAa,cAAc;AAAA,EAC7B,WAAW,wBAAwB;AACjC,iBAAa,cAAc;AAAA,EAC7B;AACA,MAAI,UAAU,SAAS,cAAc,aAAa,QAAW;AAC3D,iBAAa,WAAW,aAAa;AAAA,EACvC;AAKA,QAAM,2BAA2B,oBAAoB;AAErD,QAAM,OAAgC;AAAA,IACpC;AAAA,IACA,UAAU,SAAS;AAAA,IACnB,UAAU,cAAc,YAAY;AAAA,IACpC,eAAe;AAAA,EAAA;AAEjB,MAAI,cAAc,+BAA+B,QAAW;AAC1D,SAAK,6BAA6B,aAAa;AAAA,EACjD;AACA,MAAI,cAAc,uBAAuB,QAAW;AAClD,SAAK,qBAAqB,aAAa;AAAA,EACzC;AAEA,SAAO,EAAE,MAAM,OAAO,yBAAA;AACxB;AAQA,SAAS,UACP,eACA,QACc;AACd,MAAI,cAAe,QAAO;AAC1B,MAAI,CAAC,OAAQ,QAAO;AACpB,UAAQ,QAAA;AAAA,IACN,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AACH,aAAO;AAAA,IACT;AACE,aAAO;AAAA,EAAA;AAEb;AAEO,SAAS,eACd,OACA,YACQ;AACR,UAAQ,OAAA;AAAA,IACN,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AACH,aAAO;AAAA,IACT,KAAK;AAEH,aAAO,kBAAkB,UAAU;AAAA,IACrC,KAAK;AAOH,aAAO,eAAe,MAClB,gBACA,mBAAmB,UAAU;AAAA,IACnC,KAAK;AACH,aAAO,eAAe,MAClB,uBACA,mBAAmB,UAAU;AAAA,EAAA;AAEvC;AAeO,SAAS,iBACd,OACA,QACA,QAC2B;AAC3B,SAAO,IAAI,kBAAkB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAC3D;AAQO,SAAS,WACd,OACA,QAC2B;AAC3B,QAAM,SAAS,qBAAA;AACf,SAAO,iBAAiB,OAAO,QAAQ,MAAM;AAC/C;"}
|
|
@@ -1,128 +0,0 @@
|
|
|
1
|
-
import { BaseVideoAdapter, DurationOptions } from '@tanstack/ai/adapters';
|
|
2
|
-
import { VideoGenerationOptions, VideoJobResult, VideoStatusResult, VideoUrlResult } from '@tanstack/ai';
|
|
3
|
-
import { GrokVideoModel } from '../model-meta.js';
|
|
4
|
-
import { GrokVideoModelDurationByName, GrokVideoModelInputModalitiesByName, GrokVideoModelProviderOptionsByName, GrokVideoModelSizeByName, GrokVideoProviderOptions } from '../video/video-provider-options.js';
|
|
5
|
-
import { GrokClientConfig } from '../utils.js';
|
|
6
|
-
/**
|
|
7
|
-
* Configuration for Grok video adapter.
|
|
8
|
-
*
|
|
9
|
-
* @experimental Video generation is an experimental feature and may change.
|
|
10
|
-
*/
|
|
11
|
-
export interface GrokVideoConfig extends GrokClientConfig {
|
|
12
|
-
}
|
|
13
|
-
/**
|
|
14
|
-
* Grok Video Generation Adapter (xAI Imagine API)
|
|
15
|
-
*
|
|
16
|
-
* Tree-shakeable adapter for the grok-imagine video models using the
|
|
17
|
-
* async jobs/polling architecture: create a generation request, poll it,
|
|
18
|
-
* then read the completed video URL.
|
|
19
|
-
*
|
|
20
|
-
* `grok-imagine-video` (v1.0) supports text-to-video and image-to-video.
|
|
21
|
-
* `grok-imagine-video-1.5` is image-to-video only — every request needs an
|
|
22
|
-
* image prompt part as the starting frame, and the adapter rejects a
|
|
23
|
-
* text-only prompt with a clear error rather than a raw API 400.
|
|
24
|
-
*
|
|
25
|
-
* The Imagine video endpoints are not part of the OpenAI SDK surface (and
|
|
26
|
-
* xAI rejects the SDK's multipart paths), so requests are plain JSON calls
|
|
27
|
-
* issued with the configured `fetch` (or the global one).
|
|
28
|
-
*
|
|
29
|
-
* @experimental Video generation is an experimental feature and may change.
|
|
30
|
-
*
|
|
31
|
-
* Features:
|
|
32
|
-
* - Async job-based video generation (1–15 second clips with audio)
|
|
33
|
-
* - Aspect-ratio sizing via the "aspectRatio_resolution" size template
|
|
34
|
-
* (e.g. '16:9_720p'), consistent with the grok-imagine image models
|
|
35
|
-
* - Image-to-video via an `image` prompt part (starting frame URL or data URI)
|
|
36
|
-
* - Usage reporting: billed seconds (`unitsBilled`) and exact cost
|
|
37
|
-
*/
|
|
38
|
-
export declare class GrokVideoAdapter<TModel extends GrokVideoModel> extends BaseVideoAdapter<TModel, GrokVideoProviderOptions, GrokVideoModelProviderOptionsByName, GrokVideoModelSizeByName, GrokVideoModelInputModalitiesByName, GrokVideoModelDurationByName> {
|
|
39
|
-
readonly name: "grok";
|
|
40
|
-
private readonly clientConfig;
|
|
41
|
-
constructor(config: GrokVideoConfig, model: TModel);
|
|
42
|
-
private get fetch();
|
|
43
|
-
private request;
|
|
44
|
-
/**
|
|
45
|
-
* Reads the error message out of an Imagine API error body
|
|
46
|
-
* (`{"code": "...", "error": "..."}`), falling back to the raw text.
|
|
47
|
-
*/
|
|
48
|
-
private errorMessage;
|
|
49
|
-
createVideoJob(options: VideoGenerationOptions<GrokVideoProviderOptions, GrokVideoModelSizeByName[TModel], GrokVideoModelDurationByName[TModel]>): Promise<VideoJobResult>;
|
|
50
|
-
private retrieveJob;
|
|
51
|
-
getVideoStatus(jobId: string): Promise<VideoStatusResult>;
|
|
52
|
-
getVideoUrl(jobId: string): Promise<VideoUrlResult>;
|
|
53
|
-
/**
|
|
54
|
-
* Maps Imagine API job statuses onto the generic video status set. The
|
|
55
|
-
* API reports 'pending' while queued/generating (with a numeric
|
|
56
|
-
* `progress`), then a terminal 'done' / 'failed' / 'expired'.
|
|
57
|
-
*/
|
|
58
|
-
protected mapStatus(apiStatus: string | undefined): 'pending' | 'processing' | 'completed' | 'failed';
|
|
59
|
-
/**
|
|
60
|
-
* Both grok-imagine video models accept a continuous 1–15 integer-second
|
|
61
|
-
* range. Consumers can use this to render UI without provider knowledge.
|
|
62
|
-
*/
|
|
63
|
-
availableDurations(): DurationOptions<GrokVideoModelDurationByName[TModel]>;
|
|
64
|
-
/**
|
|
65
|
-
* Coerce a raw seconds value to the closest valid duration (clamped to
|
|
66
|
-
* [1, 15] and rounded to whole seconds).
|
|
67
|
-
*/
|
|
68
|
-
snapDuration(seconds: number): GrokVideoModelDurationByName[TModel] | undefined;
|
|
69
|
-
}
|
|
70
|
-
/**
|
|
71
|
-
* Creates a Grok video adapter with an explicit API key.
|
|
72
|
-
* Type resolution happens here at the call site.
|
|
73
|
-
*
|
|
74
|
-
* @experimental Video generation is an experimental feature and may change.
|
|
75
|
-
*
|
|
76
|
-
* @param model - The model name (e.g., 'grok-imagine-video')
|
|
77
|
-
* @param apiKey - Your xAI API key
|
|
78
|
-
* @param config - Optional additional configuration
|
|
79
|
-
* @returns Configured Grok video adapter instance with resolved types
|
|
80
|
-
*
|
|
81
|
-
* @example
|
|
82
|
-
* ```typescript
|
|
83
|
-
* // grok-imagine-video (v1.0) supports text-to-video.
|
|
84
|
-
* const adapter = createGrokVideo('grok-imagine-video', 'xai-...');
|
|
85
|
-
*
|
|
86
|
-
* const { jobId } = await generateVideo({
|
|
87
|
-
* adapter,
|
|
88
|
-
* prompt: 'A beautiful sunset over the ocean',
|
|
89
|
-
* size: '16:9_720p',
|
|
90
|
-
* duration: 5
|
|
91
|
-
* });
|
|
92
|
-
* ```
|
|
93
|
-
*/
|
|
94
|
-
export declare function createGrokVideo<TModel extends GrokVideoModel>(model: TModel, apiKey: string, config?: Omit<GrokVideoConfig, 'apiKey'>): GrokVideoAdapter<TModel>;
|
|
95
|
-
/**
|
|
96
|
-
* Creates a Grok video adapter with automatic API key detection from environment variables.
|
|
97
|
-
* Type resolution happens here at the call site.
|
|
98
|
-
*
|
|
99
|
-
* Looks for `XAI_API_KEY` in:
|
|
100
|
-
* - `process.env` (Node.js)
|
|
101
|
-
* - `window.env` (Browser with injected env)
|
|
102
|
-
*
|
|
103
|
-
* @experimental Video generation is an experimental feature and may change.
|
|
104
|
-
*
|
|
105
|
-
* @param model - The model name (e.g., 'grok-imagine-video-1.5')
|
|
106
|
-
* @param config - Optional configuration (excluding apiKey which is auto-detected)
|
|
107
|
-
* @returns Configured Grok video adapter instance with resolved types
|
|
108
|
-
* @throws Error if XAI_API_KEY is not found in environment
|
|
109
|
-
*
|
|
110
|
-
* @example
|
|
111
|
-
* ```typescript
|
|
112
|
-
* // Automatically uses XAI_API_KEY from environment
|
|
113
|
-
* const adapter = grokVideo('grok-imagine-video-1.5');
|
|
114
|
-
*
|
|
115
|
-
* // Image-to-video only: the prompt must carry a starting-frame image part.
|
|
116
|
-
* const { jobId } = await generateVideo({
|
|
117
|
-
* adapter,
|
|
118
|
-
* prompt: [
|
|
119
|
-
* { type: 'text', content: 'Make the cat start playing the piano' },
|
|
120
|
-
* { type: 'image', source: { type: 'url', value: 'https://example.com/cat.png' } },
|
|
121
|
-
* ],
|
|
122
|
-
* });
|
|
123
|
-
*
|
|
124
|
-
* // Poll for status
|
|
125
|
-
* const status = await getVideoJobStatus({ adapter, jobId });
|
|
126
|
-
* ```
|
|
127
|
-
*/
|
|
128
|
-
export declare function grokVideo<TModel extends GrokVideoModel>(model: TModel, config?: Omit<GrokVideoConfig, 'apiKey'>): GrokVideoAdapter<TModel>;
|