@tanstack/ai-grok 0.11.3 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.d.ts +19 -4
- package/dist/esm/adapters/image.js +124 -5
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/image/image-provider-options.d.ts +67 -0
- package/dist/esm/image/image-provider-options.js +37 -0
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/model-meta.d.ts +1 -1
- package/dist/esm/model-meta.js +11 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/package.json +15 -7
- package/src/adapters/image.ts +190 -9
- package/src/image/image-provider-options.ts +134 -0
- package/src/model-meta.ts +38 -1
|
@@ -2,7 +2,7 @@ import { default as OpenAI } from 'openai';
|
|
|
2
2
|
import { BaseImageAdapter } from '@tanstack/ai/adapters';
|
|
3
3
|
import { ImageGenerationOptions, ImageGenerationResult } from '@tanstack/ai';
|
|
4
4
|
import { GrokImageModel } from '../model-meta.js';
|
|
5
|
-
import { GrokImageModelProviderOptionsByName, GrokImageModelSizeByName, GrokImageProviderOptions } from '../image/image-provider-options.js';
|
|
5
|
+
import { GrokImageModelInputModalitiesByName, GrokImageModelProviderOptionsByName, GrokImageModelSizeByName, GrokImageProviderOptions } from '../image/image-provider-options.js';
|
|
6
6
|
import { GrokClientConfig } from '../utils.js';
|
|
7
7
|
/**
|
|
8
8
|
* Configuration for Grok image adapter
|
|
@@ -13,19 +13,34 @@ export interface GrokImageConfig extends GrokClientConfig {
|
|
|
13
13
|
* Grok Image Generation Adapter
|
|
14
14
|
*
|
|
15
15
|
* Tree-shakeable adapter for Grok image generation functionality.
|
|
16
|
-
* Supports grok-2-image-1212 model
|
|
16
|
+
* Supports the legacy grok-2-image-1212 model (text-to-image via the
|
|
17
|
+
* OpenAI-compat endpoint) and the grok-imagine image models, which also
|
|
18
|
+
* accept image prompt parts for image-conditioned generation via xAI's
|
|
19
|
+
* `/v1/images/edits` endpoint (up to 3 source images).
|
|
17
20
|
*
|
|
18
21
|
* Features:
|
|
19
22
|
* - Model-specific type-safe provider options
|
|
20
|
-
* - Size validation per model
|
|
23
|
+
* - Size / aspect-ratio validation per model
|
|
21
24
|
* - Number of images validation
|
|
22
25
|
*/
|
|
23
|
-
export declare class GrokImageAdapter<TModel extends GrokImageModel> extends BaseImageAdapter<TModel, GrokImageProviderOptions, GrokImageModelProviderOptionsByName, GrokImageModelSizeByName> {
|
|
26
|
+
export declare class GrokImageAdapter<TModel extends GrokImageModel> extends BaseImageAdapter<TModel, GrokImageProviderOptions, GrokImageModelProviderOptionsByName, GrokImageModelSizeByName, GrokImageModelInputModalitiesByName> {
|
|
24
27
|
readonly kind: "image";
|
|
25
28
|
readonly name: "grok";
|
|
26
29
|
protected client: OpenAI;
|
|
30
|
+
private readonly clientConfig;
|
|
27
31
|
constructor(config: GrokImageConfig, model: TModel);
|
|
28
32
|
generateImages(options: ImageGenerationOptions<GrokImageProviderOptions>): Promise<ImageGenerationResult>;
|
|
33
|
+
/**
|
|
34
|
+
* Image-conditioned generation via xAI's Imagine API.
|
|
35
|
+
*
|
|
36
|
+
* The `/v1/images/edits` endpoint takes `application/json` (the OpenAI
|
|
37
|
+
* SDK's `images.edit()` sends `multipart/form-data`, which xAI rejects),
|
|
38
|
+
* so this path issues the request directly. One input is sent as
|
|
39
|
+
* `image: { url }`; multiple inputs (up to 3) as `images: [{ url }, ...]`,
|
|
40
|
+
* addressed by xAI in the order they are sent. The prompt text is sent
|
|
41
|
+
* verbatim — no referencing markers are injected.
|
|
42
|
+
*/
|
|
43
|
+
private editImages;
|
|
29
44
|
}
|
|
30
45
|
/**
|
|
31
46
|
* Creates a Grok image adapter with explicit API key.
|
|
@@ -1,29 +1,63 @@
|
|
|
1
1
|
import OpenAI from "openai";
|
|
2
|
+
import { resolveMediaPrompt } from "@tanstack/ai";
|
|
2
3
|
import { BaseImageAdapter } from "@tanstack/ai/adapters";
|
|
3
4
|
import { toRunErrorPayload } from "@tanstack/ai/adapter-internals";
|
|
4
5
|
import { buildImagesUsage } from "@tanstack/openai-base";
|
|
5
6
|
import { generateId } from "@tanstack/ai-utils";
|
|
6
7
|
import { withGrokDefaults, getGrokApiKeyFromEnv } from "../utils/client.js";
|
|
7
|
-
import { validatePrompt, validateImageSize, validateNumberOfImages } from "../image/image-provider-options.js";
|
|
8
|
+
import { isGrokImagineImageModel, validatePrompt, validateImageSize, validateNumberOfImages, parseGrokImagineSize } from "../image/image-provider-options.js";
|
|
9
|
+
const MAX_EDIT_IMAGES = 3;
|
|
10
|
+
function imagineSizeParams(size) {
|
|
11
|
+
if (!size) return {};
|
|
12
|
+
const parsed = parseGrokImagineSize(size);
|
|
13
|
+
if (!parsed) return {};
|
|
14
|
+
return {
|
|
15
|
+
aspect_ratio: parsed.aspectRatio,
|
|
16
|
+
...parsed.resolution !== void 0 && { resolution: parsed.resolution }
|
|
17
|
+
};
|
|
18
|
+
}
|
|
19
|
+
function imagePartToUrl(part) {
|
|
20
|
+
if (part.source.type === "url") return part.source.value;
|
|
21
|
+
return `data:${part.source.mimeType};base64,${part.source.value}`;
|
|
22
|
+
}
|
|
8
23
|
class GrokImageAdapter extends BaseImageAdapter {
|
|
9
24
|
kind = "image";
|
|
10
25
|
name = "grok";
|
|
11
26
|
client;
|
|
27
|
+
clientConfig;
|
|
12
28
|
constructor(config, model) {
|
|
13
29
|
super(model, {});
|
|
14
|
-
this.
|
|
30
|
+
this.clientConfig = withGrokDefaults(config);
|
|
31
|
+
this.client = new OpenAI(this.clientConfig);
|
|
15
32
|
}
|
|
16
33
|
async generateImages(options) {
|
|
17
|
-
const { model,
|
|
34
|
+
const { model, numberOfImages, size, modelOptions } = options;
|
|
35
|
+
const resolved = resolveMediaPrompt(options.prompt);
|
|
36
|
+
const prompt = resolved.text;
|
|
37
|
+
if (resolved.videos.length > 0 || resolved.audios.length > 0) {
|
|
38
|
+
throw new Error(
|
|
39
|
+
`grok.generateImages does not support video / audio prompt parts on model ${model}.`
|
|
40
|
+
);
|
|
41
|
+
}
|
|
42
|
+
if (resolved.images.length > 0) {
|
|
43
|
+
if (!isGrokImagineImageModel(model)) {
|
|
44
|
+
throw new Error(
|
|
45
|
+
`grok: model "${model}" does not support image prompt parts. Image-conditioned generation requires an Imagine API model ('grok-imagine-image' or 'grok-imagine-image-quality').`
|
|
46
|
+
);
|
|
47
|
+
}
|
|
48
|
+
return await this.editImages(options, resolved);
|
|
49
|
+
}
|
|
18
50
|
validatePrompt({ prompt });
|
|
19
51
|
validateImageSize(model, size);
|
|
20
52
|
validateNumberOfImages(model, numberOfImages);
|
|
21
|
-
const
|
|
53
|
+
const isImagine = isGrokImagineImageModel(model);
|
|
22
54
|
const request = {
|
|
23
55
|
model,
|
|
24
56
|
prompt,
|
|
25
57
|
n: numberOfImages ?? 1,
|
|
26
|
-
...
|
|
58
|
+
...isImagine ? imagineSizeParams(size) : size !== void 0 && {
|
|
59
|
+
size
|
|
60
|
+
},
|
|
27
61
|
stream: false,
|
|
28
62
|
...modelOptions
|
|
29
63
|
};
|
|
@@ -70,6 +104,91 @@ class GrokImageAdapter extends BaseImageAdapter {
|
|
|
70
104
|
throw error;
|
|
71
105
|
}
|
|
72
106
|
}
|
|
107
|
+
/**
|
|
108
|
+
* Image-conditioned generation via xAI's Imagine API.
|
|
109
|
+
*
|
|
110
|
+
* The `/v1/images/edits` endpoint takes `application/json` (the OpenAI
|
|
111
|
+
* SDK's `images.edit()` sends `multipart/form-data`, which xAI rejects),
|
|
112
|
+
* so this path issues the request directly. One input is sent as
|
|
113
|
+
* `image: { url }`; multiple inputs (up to 3) as `images: [{ url }, ...]`,
|
|
114
|
+
* addressed by xAI in the order they are sent. The prompt text is sent
|
|
115
|
+
* verbatim — no referencing markers are injected.
|
|
116
|
+
*/
|
|
117
|
+
async editImages(options, resolved) {
|
|
118
|
+
const { model, numberOfImages, size, modelOptions, logger } = options;
|
|
119
|
+
const prompt = resolved.text;
|
|
120
|
+
const imageInputs = resolved.images;
|
|
121
|
+
const unsupportedRole = imageInputs.find(
|
|
122
|
+
(part) => part.metadata?.role === "mask" || part.metadata?.role === "control"
|
|
123
|
+
);
|
|
124
|
+
if (unsupportedRole) {
|
|
125
|
+
throw new Error(
|
|
126
|
+
`grok: the Imagine API has no ${unsupportedRole.metadata?.role} input; only source/reference images are supported.`
|
|
127
|
+
);
|
|
128
|
+
}
|
|
129
|
+
if (imageInputs.length > MAX_EDIT_IMAGES) {
|
|
130
|
+
throw new Error(
|
|
131
|
+
`grok: model "${model}" accepts at most ${MAX_EDIT_IMAGES} source images; received ${imageInputs.length}.`
|
|
132
|
+
);
|
|
133
|
+
}
|
|
134
|
+
validatePrompt({ prompt });
|
|
135
|
+
validateImageSize(model, size);
|
|
136
|
+
validateNumberOfImages(model, numberOfImages);
|
|
137
|
+
const urls = imageInputs.map((part) => imagePartToUrl(part));
|
|
138
|
+
const request = {
|
|
139
|
+
model,
|
|
140
|
+
prompt,
|
|
141
|
+
...urls.length === 1 ? { image: { url: urls[0] } } : { images: urls.map((url) => ({ url })) },
|
|
142
|
+
...numberOfImages !== void 0 && { n: numberOfImages },
|
|
143
|
+
...imagineSizeParams(size),
|
|
144
|
+
...modelOptions
|
|
145
|
+
};
|
|
146
|
+
try {
|
|
147
|
+
logger.request(
|
|
148
|
+
`activity=image provider=${this.name} model=${model} edit images=${urls.length}`,
|
|
149
|
+
{ provider: this.name, model }
|
|
150
|
+
);
|
|
151
|
+
const response = await fetch(
|
|
152
|
+
`${this.clientConfig.baseURL}/images/edits`,
|
|
153
|
+
{
|
|
154
|
+
method: "POST",
|
|
155
|
+
headers: {
|
|
156
|
+
"Content-Type": "application/json",
|
|
157
|
+
Authorization: `Bearer ${this.clientConfig.apiKey}`
|
|
158
|
+
},
|
|
159
|
+
body: JSON.stringify(request)
|
|
160
|
+
}
|
|
161
|
+
);
|
|
162
|
+
if (!response.ok) {
|
|
163
|
+
const body = await response.text();
|
|
164
|
+
throw new Error(
|
|
165
|
+
`grok: image edit request failed (${response.status} ${response.statusText}): ${body}`
|
|
166
|
+
);
|
|
167
|
+
}
|
|
168
|
+
const result = await response.json();
|
|
169
|
+
const images = (result.data ?? []).flatMap(
|
|
170
|
+
(item) => {
|
|
171
|
+
if (item.b64_json) return [{ b64Json: item.b64_json }];
|
|
172
|
+
if (item.url) return [{ url: item.url }];
|
|
173
|
+
return [];
|
|
174
|
+
}
|
|
175
|
+
);
|
|
176
|
+
if (images.length === 0) {
|
|
177
|
+
throw new Error("grok: image edit response contained no images");
|
|
178
|
+
}
|
|
179
|
+
return {
|
|
180
|
+
id: generateId(this.name),
|
|
181
|
+
model,
|
|
182
|
+
images
|
|
183
|
+
};
|
|
184
|
+
} catch (error) {
|
|
185
|
+
logger.errors(`${this.name}.generateImages fatal`, {
|
|
186
|
+
error: toRunErrorPayload(error, `${this.name}.generateImages failed`),
|
|
187
|
+
source: `${this.name}.generateImages`
|
|
188
|
+
});
|
|
189
|
+
throw error;
|
|
190
|
+
}
|
|
191
|
+
}
|
|
73
192
|
}
|
|
74
193
|
function createGrokImage(model, apiKey, config) {
|
|
75
194
|
return new GrokImageAdapter({ apiKey, ...config }, model);
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"image.js","sources":["../../../src/adapters/image.ts"],"sourcesContent":["import OpenAI from 'openai'\nimport { BaseImageAdapter } from '@tanstack/ai/adapters'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport { buildImagesUsage } from '@tanstack/openai-base'\nimport { generateId } from '@tanstack/ai-utils'\nimport { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'\nimport {\n validateImageSize,\n validateNumberOfImages,\n validatePrompt,\n} from '../image/image-provider-options'\nimport type {\n GeneratedImage,\n ImageGenerationOptions,\n ImageGenerationResult,\n} from '@tanstack/ai'\nimport type OpenAI_SDK from 'openai'\nimport type { GrokImageModel } from '../model-meta'\nimport type {\n GrokImageModelProviderOptionsByName,\n GrokImageModelSizeByName,\n GrokImageProviderOptions,\n} from '../image/image-provider-options'\nimport type { GrokClientConfig } from '../utils'\n\n/**\n * Configuration for Grok image adapter\n */\nexport interface GrokImageConfig extends GrokClientConfig {}\n\n/**\n * Grok Image Generation Adapter\n *\n * Tree-shakeable adapter for Grok image generation functionality.\n * Supports grok-2-image-1212 model.\n *\n * Features:\n * - Model-specific type-safe provider options\n * - Size validation per model\n * - Number of images validation\n */\nexport class GrokImageAdapter<\n TModel extends GrokImageModel,\n> extends BaseImageAdapter<\n TModel,\n GrokImageProviderOptions,\n GrokImageModelProviderOptionsByName,\n GrokImageModelSizeByName\n> {\n override readonly kind = 'image' as const\n readonly name = 'grok' as const\n\n protected client: OpenAI\n\n constructor(config: GrokImageConfig, model: TModel) {\n super(model, {})\n this.client = new OpenAI(withGrokDefaults(config))\n }\n\n async generateImages(\n options: ImageGenerationOptions<GrokImageProviderOptions>,\n ): Promise<ImageGenerationResult> {\n const { model, prompt, numberOfImages, size, modelOptions } = options\n\n validatePrompt({ prompt, model })\n validateImageSize(model, size)\n validateNumberOfImages(model, numberOfImages)\n\n const resolvedSize = size as OpenAI_SDK.Images.ImageGenerateParams['size']\n const request: OpenAI_SDK.Images.ImageGenerateParamsNonStreaming = {\n model,\n prompt,\n n: numberOfImages ?? 1,\n ...(resolvedSize !== undefined && { size: resolvedSize }),\n stream: false,\n ...modelOptions,\n }\n\n try {\n options.logger.request(\n `activity=image provider=${this.name} model=${model} n=${request.n ?? 1} size=${request.size ?? 'default'}`,\n { provider: this.name, model },\n )\n const response = await this.client.images.generate(request)\n\n const images: Array<GeneratedImage> = (response.data ?? []).flatMap(\n (item): Array<GeneratedImage> => {\n const revisedPrompt = item.revised_prompt\n if (item.b64_json) {\n return [\n {\n b64Json: item.b64_json,\n ...(revisedPrompt !== undefined && { revisedPrompt }),\n },\n ]\n }\n if (item.url) {\n return [\n {\n url: item.url,\n ...(revisedPrompt !== undefined && { revisedPrompt }),\n },\n ]\n }\n return []\n },\n )\n\n const usage = buildImagesUsage(response.usage)\n\n return {\n id: generateId(this.name),\n model,\n images,\n ...(usage ? { usage } : {}),\n }\n } catch (error: unknown) {\n options.logger.errors(`${this.name}.generateImages fatal`, {\n error: toRunErrorPayload(error, `${this.name}.generateImages failed`),\n source: `${this.name}.generateImages`,\n })\n throw error\n }\n }\n}\n\n/**\n * Creates a Grok image adapter with explicit API key.\n * Type resolution happens here at the call site.\n *\n * @param model - The model name (e.g., 'grok-2-image-1212')\n * @param apiKey - Your xAI API key\n * @param config - Optional additional configuration\n * @returns Configured Grok image adapter instance with resolved types\n *\n * @example\n * ```typescript\n * const adapter = createGrokImage('grok-2-image-1212', \"xai-...\");\n *\n * const result = await generateImage({\n * adapter,\n * prompt: 'A cute baby sea otter'\n * });\n * ```\n */\nexport function createGrokImage<TModel extends GrokImageModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GrokImageConfig, 'apiKey'>,\n): GrokImageAdapter<TModel> {\n return new GrokImageAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Grok image adapter with automatic API key detection from environment variables.\n * Type resolution happens here at the call site.\n *\n * Looks for `XAI_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (Browser with injected env)\n *\n * @param model - The model name (e.g., 'grok-2-image-1212')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Grok image adapter instance with resolved types\n * @throws Error if XAI_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * // Automatically uses XAI_API_KEY from environment\n * const adapter = grokImage('grok-2-image-1212');\n *\n * const result = await generateImage({\n * adapter,\n * prompt: 'A beautiful sunset over mountains'\n * });\n * ```\n */\nexport function grokImage<TModel extends GrokImageModel>(\n model: TModel,\n config?: Omit<GrokImageConfig, 'apiKey'>,\n): GrokImageAdapter<TModel> {\n const apiKey = getGrokApiKeyFromEnv()\n return createGrokImage(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;;;;;AAyCO,MAAM,yBAEH,iBAKR;AAAA,EACkB,OAAO;AAAA,EAChB,OAAO;AAAA,EAEN;AAAA,EAEV,YAAY,QAAyB,OAAe;AAClD,UAAM,OAAO,EAAE;AACf,SAAK,SAAS,IAAI,OAAO,iBAAiB,MAAM,CAAC;AAAA,EACnD;AAAA,EAEA,MAAM,eACJ,SACgC;AAChC,UAAM,EAAE,OAAO,QAAQ,gBAAgB,MAAM,iBAAiB;AAE9D,mBAAe,EAAE,OAAc,CAAC;AAChC,sBAAkB,OAAO,IAAI;AAC7B,2BAAuB,OAAO,cAAc;AAE5C,UAAM,eAAe;AACrB,UAAM,UAA6D;AAAA,MACjE;AAAA,MACA;AAAA,MACA,GAAG,kBAAkB;AAAA,MACrB,GAAI,iBAAiB,UAAa,EAAE,MAAM,aAAA;AAAA,MAC1C,QAAQ;AAAA,MACR,GAAG;AAAA,IAAA;AAGL,QAAI;AACF,cAAQ,OAAO;AAAA,QACb,2BAA2B,KAAK,IAAI,UAAU,KAAK,MAAM,QAAQ,KAAK,CAAC,SAAS,QAAQ,QAAQ,SAAS;AAAA,QACzG,EAAE,UAAU,KAAK,MAAM,MAAA;AAAA,MAAM;AAE/B,YAAM,WAAW,MAAM,KAAK,OAAO,OAAO,SAAS,OAAO;AAE1D,YAAM,UAAiC,SAAS,QAAQ,CAAA,GAAI;AAAA,QAC1D,CAAC,SAAgC;AAC/B,gBAAM,gBAAgB,KAAK;AAC3B,cAAI,KAAK,UAAU;AACjB,mBAAO;AAAA,cACL;AAAA,gBACE,SAAS,KAAK;AAAA,gBACd,GAAI,kBAAkB,UAAa,EAAE,cAAA;AAAA,cAAc;AAAA,YACrD;AAAA,UAEJ;AACA,cAAI,KAAK,KAAK;AACZ,mBAAO;AAAA,cACL;AAAA,gBACE,KAAK,KAAK;AAAA,gBACV,GAAI,kBAAkB,UAAa,EAAE,cAAA;AAAA,cAAc;AAAA,YACrD;AAAA,UAEJ;AACA,iBAAO,CAAA;AAAA,QACT;AAAA,MAAA;AAGF,YAAM,QAAQ,iBAAiB,SAAS,KAAK;AAE7C,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA;AAAA,QACA,GAAI,QAAQ,EAAE,UAAU,CAAA;AAAA,MAAC;AAAA,IAE7B,SAAS,OAAgB;AACvB,cAAQ,OAAO,OAAO,GAAG,KAAK,IAAI,yBAAyB;AAAA,QACzD,OAAO,kBAAkB,OAAO,GAAG,KAAK,IAAI,wBAAwB;AAAA,QACpE,QAAQ,GAAG,KAAK,IAAI;AAAA,MAAA,CACrB;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AAqBO,SAAS,gBACd,OACA,QACA,QAC0B;AAC1B,SAAO,IAAI,iBAAiB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAC1D;AA0BO,SAAS,UACd,OACA,QAC0B;AAC1B,QAAM,SAAS,qBAAA;AACf,SAAO,gBAAgB,OAAO,QAAQ,MAAM;AAC9C;"}
|
|
1
|
+
{"version":3,"file":"image.js","sources":["../../../src/adapters/image.ts"],"sourcesContent":["import OpenAI from 'openai'\nimport { resolveMediaPrompt } from '@tanstack/ai'\nimport { BaseImageAdapter } from '@tanstack/ai/adapters'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport { buildImagesUsage } from '@tanstack/openai-base'\nimport { generateId } from '@tanstack/ai-utils'\nimport { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'\nimport {\n isGrokImagineImageModel,\n parseGrokImagineSize,\n validateImageSize,\n validateNumberOfImages,\n validatePrompt,\n} from '../image/image-provider-options'\nimport type {\n GeneratedImage,\n ImageGenerationOptions,\n ImageGenerationResult,\n ImagePart,\n MediaInputMetadata,\n ResolvedMediaPrompt,\n} from '@tanstack/ai'\nimport type OpenAI_SDK from 'openai'\nimport type { GrokImageModel } from '../model-meta'\nimport type {\n GrokImageModelInputModalitiesByName,\n GrokImageModelProviderOptionsByName,\n GrokImageModelSizeByName,\n GrokImageProviderOptions,\n} from '../image/image-provider-options'\nimport type { GrokClientConfig } from '../utils'\n\n/**\n * Configuration for Grok image adapter\n */\nexport interface GrokImageConfig extends GrokClientConfig {}\n\n/** Maximum source images accepted by xAI's image edit endpoint. */\nconst MAX_EDIT_IMAGES = 3\n\n/**\n * Maps the generic `size` option onto Imagine API parameters: the\n * \"aspectRatio_resolution\" template (\"16:9_2k\") splits into `aspect_ratio`\n * and optional `resolution` request fields.\n */\nfunction imagineSizeParams(size: string | undefined): {\n aspect_ratio?: string\n resolution?: string\n} {\n if (!size) return {}\n const parsed = parseGrokImagineSize(size)\n if (!parsed) return {}\n return {\n aspect_ratio: parsed.aspectRatio,\n ...(parsed.resolution !== undefined && { resolution: parsed.resolution }),\n }\n}\n\n/**\n * Convert a TanStack ImagePart to the URL string accepted by xAI's edit\n * endpoint: public URLs pass through (fetched by xAI's servers), data\n * sources become base64 data URIs.\n */\nfunction imagePartToUrl(part: ImagePart<MediaInputMetadata>): string {\n if (part.source.type === 'url') return part.source.value\n return `data:${part.source.mimeType};base64,${part.source.value}`\n}\n\n/** Response shape of xAI's `/v1/images/edits` endpoint. */\ninterface GrokImageEditResponse {\n data?: Array<{\n url?: string | null\n b64_json?: string | null\n mime_type?: string\n }>\n}\n\n/**\n * Grok Image Generation Adapter\n *\n * Tree-shakeable adapter for Grok image generation functionality.\n * Supports the legacy grok-2-image-1212 model (text-to-image via the\n * OpenAI-compat endpoint) and the grok-imagine image models, which also\n * accept image prompt parts for image-conditioned generation via xAI's\n * `/v1/images/edits` endpoint (up to 3 source images).\n *\n * Features:\n * - Model-specific type-safe provider options\n * - Size / aspect-ratio validation per model\n * - Number of images validation\n */\nexport class GrokImageAdapter<\n TModel extends GrokImageModel,\n> extends BaseImageAdapter<\n TModel,\n GrokImageProviderOptions,\n GrokImageModelProviderOptionsByName,\n GrokImageModelSizeByName,\n GrokImageModelInputModalitiesByName\n> {\n override readonly kind = 'image' as const\n readonly name = 'grok' as const\n\n protected client: OpenAI\n private readonly clientConfig: GrokImageConfig\n\n constructor(config: GrokImageConfig, model: TModel) {\n super(model, {})\n this.clientConfig = withGrokDefaults(config)\n this.client = new OpenAI(this.clientConfig)\n }\n\n async generateImages(\n options: ImageGenerationOptions<GrokImageProviderOptions>,\n ): Promise<ImageGenerationResult> {\n const { model, numberOfImages, size, modelOptions } = options\n\n const resolved = resolveMediaPrompt(options.prompt)\n const prompt = resolved.text\n\n if (resolved.videos.length > 0 || resolved.audios.length > 0) {\n throw new Error(\n `grok.generateImages does not support video / audio prompt parts on model ${model}.`,\n )\n }\n\n if (resolved.images.length > 0) {\n if (!isGrokImagineImageModel(model)) {\n throw new Error(\n `grok: model \"${model}\" does not support image prompt parts. ` +\n `Image-conditioned generation requires an Imagine API model ` +\n `('grok-imagine-image' or 'grok-imagine-image-quality').`,\n )\n }\n return await this.editImages(options, resolved)\n }\n\n validatePrompt({ prompt, model })\n validateImageSize(model, size)\n validateNumberOfImages(model, numberOfImages)\n\n // grok-imagine models are aspect-ratio sized: the generic `size` option\n // carries an \"aspectRatio_resolution\" template (e.g. '16:9_2k', like\n // Gemini native image models) and maps to the Imagine API's\n // `aspect_ratio` / `resolution` parameters instead of OpenAI-style `size`.\n const isImagine = isGrokImagineImageModel(model)\n const request = {\n model,\n prompt,\n n: numberOfImages ?? 1,\n ...(isImagine\n ? imagineSizeParams(size)\n : size !== undefined && {\n size: size,\n }),\n stream: false,\n ...modelOptions,\n } as OpenAI_SDK.Images.ImageGenerateParamsNonStreaming\n\n try {\n options.logger.request(\n `activity=image provider=${this.name} model=${model} n=${request.n ?? 1} size=${request.size ?? 'default'}`,\n { provider: this.name, model },\n )\n const response = await this.client.images.generate(request)\n\n const images: Array<GeneratedImage> = (response.data ?? []).flatMap(\n (item): Array<GeneratedImage> => {\n const revisedPrompt = item.revised_prompt\n if (item.b64_json) {\n return [\n {\n b64Json: item.b64_json,\n ...(revisedPrompt !== undefined && { revisedPrompt }),\n },\n ]\n }\n if (item.url) {\n return [\n {\n url: item.url,\n ...(revisedPrompt !== undefined && { revisedPrompt }),\n },\n ]\n }\n return []\n },\n )\n\n const usage = buildImagesUsage(response.usage)\n\n return {\n id: generateId(this.name),\n model,\n images,\n ...(usage ? { usage } : {}),\n }\n } catch (error: unknown) {\n options.logger.errors(`${this.name}.generateImages fatal`, {\n error: toRunErrorPayload(error, `${this.name}.generateImages failed`),\n source: `${this.name}.generateImages`,\n })\n throw error\n }\n }\n\n /**\n * Image-conditioned generation via xAI's Imagine API.\n *\n * The `/v1/images/edits` endpoint takes `application/json` (the OpenAI\n * SDK's `images.edit()` sends `multipart/form-data`, which xAI rejects),\n * so this path issues the request directly. One input is sent as\n * `image: { url }`; multiple inputs (up to 3) as `images: [{ url }, ...]`,\n * addressed by xAI in the order they are sent. The prompt text is sent\n * verbatim — no referencing markers are injected.\n */\n private async editImages(\n options: ImageGenerationOptions<GrokImageProviderOptions>,\n resolved: ResolvedMediaPrompt,\n ): Promise<ImageGenerationResult> {\n const { model, numberOfImages, size, modelOptions, logger } = options\n const prompt = resolved.text\n const imageInputs = resolved.images\n\n const unsupportedRole = imageInputs.find(\n (part) =>\n part.metadata?.role === 'mask' || part.metadata?.role === 'control',\n )\n if (unsupportedRole) {\n throw new Error(\n `grok: the Imagine API has no ${unsupportedRole.metadata?.role} input; ` +\n `only source/reference images are supported.`,\n )\n }\n if (imageInputs.length > MAX_EDIT_IMAGES) {\n throw new Error(\n `grok: model \"${model}\" accepts at most ${MAX_EDIT_IMAGES} source images; received ${imageInputs.length}.`,\n )\n }\n\n validatePrompt({ prompt, model })\n validateImageSize(model, size)\n validateNumberOfImages(model, numberOfImages)\n\n const urls = imageInputs.map((part) => imagePartToUrl(part))\n const request: Record<string, unknown> = {\n model,\n prompt,\n ...(urls.length === 1\n ? { image: { url: urls[0] } }\n : { images: urls.map((url) => ({ url })) }),\n ...(numberOfImages !== undefined && { n: numberOfImages }),\n ...imagineSizeParams(size),\n ...modelOptions,\n }\n\n try {\n logger.request(\n `activity=image provider=${this.name} model=${model} edit images=${urls.length}`,\n { provider: this.name, model },\n )\n\n const response = await fetch(\n `${this.clientConfig.baseURL}/images/edits`,\n {\n method: 'POST',\n headers: {\n 'Content-Type': 'application/json',\n Authorization: `Bearer ${this.clientConfig.apiKey}`,\n },\n body: JSON.stringify(request),\n },\n )\n if (!response.ok) {\n const body = await response.text()\n throw new Error(\n `grok: image edit request failed (${response.status} ${response.statusText}): ${body}`,\n )\n }\n\n const result = (await response.json()) as GrokImageEditResponse\n const images: Array<GeneratedImage> = (result.data ?? []).flatMap(\n (item): Array<GeneratedImage> => {\n if (item.b64_json) return [{ b64Json: item.b64_json }]\n if (item.url) return [{ url: item.url }]\n return []\n },\n )\n if (images.length === 0) {\n throw new Error('grok: image edit response contained no images')\n }\n\n return {\n id: generateId(this.name),\n model,\n images,\n }\n } catch (error: unknown) {\n logger.errors(`${this.name}.generateImages fatal`, {\n error: toRunErrorPayload(error, `${this.name}.generateImages failed`),\n source: `${this.name}.generateImages`,\n })\n throw error\n }\n }\n}\n\n/**\n * Creates a Grok image adapter with explicit API key.\n * Type resolution happens here at the call site.\n *\n * @param model - The model name (e.g., 'grok-2-image-1212')\n * @param apiKey - Your xAI API key\n * @param config - Optional additional configuration\n * @returns Configured Grok image adapter instance with resolved types\n *\n * @example\n * ```typescript\n * const adapter = createGrokImage('grok-2-image-1212', \"xai-...\");\n *\n * const result = await generateImage({\n * adapter,\n * prompt: 'A cute baby sea otter'\n * });\n * ```\n */\nexport function createGrokImage<TModel extends GrokImageModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GrokImageConfig, 'apiKey'>,\n): GrokImageAdapter<TModel> {\n return new GrokImageAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Grok image adapter with automatic API key detection from environment variables.\n * Type resolution happens here at the call site.\n *\n * Looks for `XAI_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (Browser with injected env)\n *\n * @param model - The model name (e.g., 'grok-2-image-1212')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Grok image adapter instance with resolved types\n * @throws Error if XAI_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * // Automatically uses XAI_API_KEY from environment\n * const adapter = grokImage('grok-2-image-1212');\n *\n * const result = await generateImage({\n * adapter,\n * prompt: 'A beautiful sunset over mountains'\n * });\n * ```\n */\nexport function grokImage<TModel extends GrokImageModel>(\n model: TModel,\n config?: Omit<GrokImageConfig, 'apiKey'>,\n): GrokImageAdapter<TModel> {\n const apiKey = getGrokApiKeyFromEnv()\n return createGrokImage(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;;;;;;AAsCA,MAAM,kBAAkB;AAOxB,SAAS,kBAAkB,MAGzB;AACA,MAAI,CAAC,KAAM,QAAO,CAAA;AAClB,QAAM,SAAS,qBAAqB,IAAI;AACxC,MAAI,CAAC,OAAQ,QAAO,CAAA;AACpB,SAAO;AAAA,IACL,cAAc,OAAO;AAAA,IACrB,GAAI,OAAO,eAAe,UAAa,EAAE,YAAY,OAAO,WAAA;AAAA,EAAW;AAE3E;AAOA,SAAS,eAAe,MAA6C;AACnE,MAAI,KAAK,OAAO,SAAS,MAAO,QAAO,KAAK,OAAO;AACnD,SAAO,QAAQ,KAAK,OAAO,QAAQ,WAAW,KAAK,OAAO,KAAK;AACjE;AAyBO,MAAM,yBAEH,iBAMR;AAAA,EACkB,OAAO;AAAA,EAChB,OAAO;AAAA,EAEN;AAAA,EACO;AAAA,EAEjB,YAAY,QAAyB,OAAe;AAClD,UAAM,OAAO,EAAE;AACf,SAAK,eAAe,iBAAiB,MAAM;AAC3C,SAAK,SAAS,IAAI,OAAO,KAAK,YAAY;AAAA,EAC5C;AAAA,EAEA,MAAM,eACJ,SACgC;AAChC,UAAM,EAAE,OAAO,gBAAgB,MAAM,iBAAiB;AAEtD,UAAM,WAAW,mBAAmB,QAAQ,MAAM;AAClD,UAAM,SAAS,SAAS;AAExB,QAAI,SAAS,OAAO,SAAS,KAAK,SAAS,OAAO,SAAS,GAAG;AAC5D,YAAM,IAAI;AAAA,QACR,4EAA4E,KAAK;AAAA,MAAA;AAAA,IAErF;AAEA,QAAI,SAAS,OAAO,SAAS,GAAG;AAC9B,UAAI,CAAC,wBAAwB,KAAK,GAAG;AACnC,cAAM,IAAI;AAAA,UACR,gBAAgB,KAAK;AAAA,QAAA;AAAA,MAIzB;AACA,aAAO,MAAM,KAAK,WAAW,SAAS,QAAQ;AAAA,IAChD;AAEA,mBAAe,EAAE,OAAc,CAAC;AAChC,sBAAkB,OAAO,IAAI;AAC7B,2BAAuB,OAAO,cAAc;AAM5C,UAAM,YAAY,wBAAwB,KAAK;AAC/C,UAAM,UAAU;AAAA,MACd;AAAA,MACA;AAAA,MACA,GAAG,kBAAkB;AAAA,MACrB,GAAI,YACA,kBAAkB,IAAI,IACtB,SAAS,UAAa;AAAA,QACpB;AAAA,MAAA;AAAA,MAEN,QAAQ;AAAA,MACR,GAAG;AAAA,IAAA;AAGL,QAAI;AACF,cAAQ,OAAO;AAAA,QACb,2BAA2B,KAAK,IAAI,UAAU,KAAK,MAAM,QAAQ,KAAK,CAAC,SAAS,QAAQ,QAAQ,SAAS;AAAA,QACzG,EAAE,UAAU,KAAK,MAAM,MAAA;AAAA,MAAM;AAE/B,YAAM,WAAW,MAAM,KAAK,OAAO,OAAO,SAAS,OAAO;AAE1D,YAAM,UAAiC,SAAS,QAAQ,CAAA,GAAI;AAAA,QAC1D,CAAC,SAAgC;AAC/B,gBAAM,gBAAgB,KAAK;AAC3B,cAAI,KAAK,UAAU;AACjB,mBAAO;AAAA,cACL;AAAA,gBACE,SAAS,KAAK;AAAA,gBACd,GAAI,kBAAkB,UAAa,EAAE,cAAA;AAAA,cAAc;AAAA,YACrD;AAAA,UAEJ;AACA,cAAI,KAAK,KAAK;AACZ,mBAAO;AAAA,cACL;AAAA,gBACE,KAAK,KAAK;AAAA,gBACV,GAAI,kBAAkB,UAAa,EAAE,cAAA;AAAA,cAAc;AAAA,YACrD;AAAA,UAEJ;AACA,iBAAO,CAAA;AAAA,QACT;AAAA,MAAA;AAGF,YAAM,QAAQ,iBAAiB,SAAS,KAAK;AAE7C,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA;AAAA,QACA,GAAI,QAAQ,EAAE,UAAU,CAAA;AAAA,MAAC;AAAA,IAE7B,SAAS,OAAgB;AACvB,cAAQ,OAAO,OAAO,GAAG,KAAK,IAAI,yBAAyB;AAAA,QACzD,OAAO,kBAAkB,OAAO,GAAG,KAAK,IAAI,wBAAwB;AAAA,QACpE,QAAQ,GAAG,KAAK,IAAI;AAAA,MAAA,CACrB;AACD,YAAM;AAAA,IACR;AAAA,EACF;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAYA,MAAc,WACZ,SACA,UACgC;AAChC,UAAM,EAAE,OAAO,gBAAgB,MAAM,cAAc,WAAW;AAC9D,UAAM,SAAS,SAAS;AACxB,UAAM,cAAc,SAAS;AAE7B,UAAM,kBAAkB,YAAY;AAAA,MAClC,CAAC,SACC,KAAK,UAAU,SAAS,UAAU,KAAK,UAAU,SAAS;AAAA,IAAA;AAE9D,QAAI,iBAAiB;AACnB,YAAM,IAAI;AAAA,QACR,gCAAgC,gBAAgB,UAAU,IAAI;AAAA,MAAA;AAAA,IAGlE;AACA,QAAI,YAAY,SAAS,iBAAiB;AACxC,YAAM,IAAI;AAAA,QACR,gBAAgB,KAAK,qBAAqB,eAAe,4BAA4B,YAAY,MAAM;AAAA,MAAA;AAAA,IAE3G;AAEA,mBAAe,EAAE,OAAc,CAAC;AAChC,sBAAkB,OAAO,IAAI;AAC7B,2BAAuB,OAAO,cAAc;AAE5C,UAAM,OAAO,YAAY,IAAI,CAAC,SAAS,eAAe,IAAI,CAAC;AAC3D,UAAM,UAAmC;AAAA,MACvC;AAAA,MACA;AAAA,MACA,GAAI,KAAK,WAAW,IAChB,EAAE,OAAO,EAAE,KAAK,KAAK,CAAC,EAAA,MACtB,EAAE,QAAQ,KAAK,IAAI,CAAC,SAAS,EAAE,IAAA,EAAM,EAAA;AAAA,MACzC,GAAI,mBAAmB,UAAa,EAAE,GAAG,eAAA;AAAA,MACzC,GAAG,kBAAkB,IAAI;AAAA,MACzB,GAAG;AAAA,IAAA;AAGL,QAAI;AACF,aAAO;AAAA,QACL,2BAA2B,KAAK,IAAI,UAAU,KAAK,gBAAgB,KAAK,MAAM;AAAA,QAC9E,EAAE,UAAU,KAAK,MAAM,MAAA;AAAA,MAAM;AAG/B,YAAM,WAAW,MAAM;AAAA,QACrB,GAAG,KAAK,aAAa,OAAO;AAAA,QAC5B;AAAA,UACE,QAAQ;AAAA,UACR,SAAS;AAAA,YACP,gBAAgB;AAAA,YAChB,eAAe,UAAU,KAAK,aAAa,MAAM;AAAA,UAAA;AAAA,UAEnD,MAAM,KAAK,UAAU,OAAO;AAAA,QAAA;AAAA,MAC9B;AAEF,UAAI,CAAC,SAAS,IAAI;AAChB,cAAM,OAAO,MAAM,SAAS,KAAA;AAC5B,cAAM,IAAI;AAAA,UACR,oCAAoC,SAAS,MAAM,IAAI,SAAS,UAAU,MAAM,IAAI;AAAA,QAAA;AAAA,MAExF;AAEA,YAAM,SAAU,MAAM,SAAS,KAAA;AAC/B,YAAM,UAAiC,OAAO,QAAQ,CAAA,GAAI;AAAA,QACxD,CAAC,SAAgC;AAC/B,cAAI,KAAK,SAAU,QAAO,CAAC,EAAE,SAAS,KAAK,UAAU;AACrD,cAAI,KAAK,IAAK,QAAO,CAAC,EAAE,KAAK,KAAK,KAAK;AACvC,iBAAO,CAAA;AAAA,QACT;AAAA,MAAA;AAEF,UAAI,OAAO,WAAW,GAAG;AACvB,cAAM,IAAI,MAAM,+CAA+C;AAAA,MACjE;AAEA,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA;AAAA,MAAA;AAAA,IAEJ,SAAS,OAAgB;AACvB,aAAO,OAAO,GAAG,KAAK,IAAI,yBAAyB;AAAA,QACjD,OAAO,kBAAkB,OAAO,GAAG,KAAK,IAAI,wBAAwB;AAAA,QACpE,QAAQ,GAAG,KAAK,IAAI;AAAA,MAAA,CACrB;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AAqBO,SAAS,gBACd,OACA,QACA,QAC0B;AAC1B,SAAO,IAAI,iBAAiB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAC1D;AA0BO,SAAS,UACd,OACA,QAC0B;AAC1B,QAAM,SAAS,qBAAA;AACf,SAAO,gBAAgB,OAAO,QAAQ,MAAM;AAC9C;"}
|
|
@@ -8,6 +8,38 @@
|
|
|
8
8
|
* Supported sizes for grok-2-image-1212 model
|
|
9
9
|
*/
|
|
10
10
|
export type GrokImageSize = '1024x1024' | '1536x1024' | '1024x1536';
|
|
11
|
+
/**
|
|
12
|
+
* Aspect ratios accepted by the grok-imagine image models.
|
|
13
|
+
*/
|
|
14
|
+
export type GrokImagineAspectRatio = '1:1' | '3:4' | '4:3' | '9:16' | '16:9' | '2:3' | '3:2' | '9:19.5' | '19.5:9' | '9:20' | '20:9' | '1:2' | '2:1' | 'auto';
|
|
15
|
+
/**
|
|
16
|
+
* Resolution tiers for the grok-imagine image models.
|
|
17
|
+
*/
|
|
18
|
+
export type GrokImagineResolution = '1k' | '2k';
|
|
19
|
+
/**
|
|
20
|
+
* Size strings for grok-imagine image models. The Imagine API is
|
|
21
|
+
* aspect-ratio based rather than pixel-size based; like Gemini's native
|
|
22
|
+
* image models, the generic `size` option uses an
|
|
23
|
+
* `aspectRatio_resolution` template ("16:9_2k") — the resolution suffix is
|
|
24
|
+
* optional ("16:9" uses the API default of 1k).
|
|
25
|
+
*/
|
|
26
|
+
export type GrokImagineImageSize = GrokImagineAspectRatio | `${GrokImagineAspectRatio}_${GrokImagineResolution}`;
|
|
27
|
+
/**
|
|
28
|
+
* Models served by xAI's Imagine API. They are aspect-ratio sized and
|
|
29
|
+
* support image-conditioned generation via `/v1/images/edits`; the legacy
|
|
30
|
+
* grok-2-image-1212 model is pixel-sized and text-to-image only.
|
|
31
|
+
*/
|
|
32
|
+
export declare function isGrokImagineImageModel(model: string): boolean;
|
|
33
|
+
/**
|
|
34
|
+
* Parses a grok-imagine size string into its components.
|
|
35
|
+
* Format: "aspectRatio" or "aspectRatio_resolution",
|
|
36
|
+
* e.g. "16:9_2k" → { aspectRatio: "16:9", resolution: "2k" }.
|
|
37
|
+
* Returns undefined when the string doesn't match the template.
|
|
38
|
+
*/
|
|
39
|
+
export declare function parseGrokImagineSize(size: string): {
|
|
40
|
+
aspectRatio: string;
|
|
41
|
+
resolution?: string;
|
|
42
|
+
} | undefined;
|
|
11
43
|
/**
|
|
12
44
|
* Base provider options for Grok image models
|
|
13
45
|
*/
|
|
@@ -34,17 +66,52 @@ export interface GrokImageProviderOptions extends GrokImageBaseProviderOptions {
|
|
|
34
66
|
*/
|
|
35
67
|
response_format?: 'url' | 'b64_json';
|
|
36
68
|
}
|
|
69
|
+
/**
|
|
70
|
+
* Provider options for the grok-imagine image models (generation and
|
|
71
|
+
* image-conditioned editing via xAI's Imagine API).
|
|
72
|
+
*/
|
|
73
|
+
export interface GrokImagineImageProviderOptions extends GrokImageBaseProviderOptions {
|
|
74
|
+
/**
|
|
75
|
+
* The format in which generated images are returned.
|
|
76
|
+
* @default 'url'
|
|
77
|
+
*/
|
|
78
|
+
response_format?: 'url' | 'b64_json';
|
|
79
|
+
/**
|
|
80
|
+
* Output resolution.
|
|
81
|
+
* @default '1k'
|
|
82
|
+
*/
|
|
83
|
+
resolution?: '1k' | '2k';
|
|
84
|
+
/**
|
|
85
|
+
* Processing tier for the request.
|
|
86
|
+
* @default 'default'
|
|
87
|
+
*/
|
|
88
|
+
service_tier?: 'default' | 'priority';
|
|
89
|
+
}
|
|
37
90
|
/**
|
|
38
91
|
* Type-only map from model name to its specific provider options.
|
|
39
92
|
*/
|
|
40
93
|
export type GrokImageModelProviderOptionsByName = {
|
|
41
94
|
'grok-2-image-1212': GrokImageProviderOptions;
|
|
95
|
+
'grok-imagine-image': GrokImagineImageProviderOptions;
|
|
96
|
+
'grok-imagine-image-quality': GrokImagineImageProviderOptions;
|
|
42
97
|
};
|
|
43
98
|
/**
|
|
44
99
|
* Type-only map from model name to its supported sizes.
|
|
45
100
|
*/
|
|
46
101
|
export type GrokImageModelSizeByName = {
|
|
47
102
|
'grok-2-image-1212': GrokImageSize;
|
|
103
|
+
'grok-imagine-image': GrokImagineImageSize;
|
|
104
|
+
'grok-imagine-image-quality': GrokImagineImageSize;
|
|
105
|
+
};
|
|
106
|
+
/**
|
|
107
|
+
* Per-model prompt input modalities. Imagine API models accept image parts
|
|
108
|
+
* in the prompt (routed to `/v1/images/edits`, up to 3 images, addressed by
|
|
109
|
+
* xAI in request order); grok-2-image is text-to-image only.
|
|
110
|
+
*/
|
|
111
|
+
export type GrokImageModelInputModalitiesByName = {
|
|
112
|
+
'grok-2-image-1212': readonly [];
|
|
113
|
+
'grok-imagine-image': readonly ['image'];
|
|
114
|
+
'grok-imagine-image-quality': readonly ['image'];
|
|
48
115
|
};
|
|
49
116
|
/**
|
|
50
117
|
* Internal options interface for validation
|
|
@@ -1,5 +1,40 @@
|
|
|
1
|
+
const GROK_IMAGINE_ASPECT_RATIOS = [
|
|
2
|
+
"1:1",
|
|
3
|
+
"3:4",
|
|
4
|
+
"4:3",
|
|
5
|
+
"9:16",
|
|
6
|
+
"16:9",
|
|
7
|
+
"2:3",
|
|
8
|
+
"3:2",
|
|
9
|
+
"9:19.5",
|
|
10
|
+
"19.5:9",
|
|
11
|
+
"9:20",
|
|
12
|
+
"20:9",
|
|
13
|
+
"1:2",
|
|
14
|
+
"2:1",
|
|
15
|
+
"auto"
|
|
16
|
+
];
|
|
17
|
+
const GROK_IMAGINE_RESOLUTIONS = ["1k", "2k"];
|
|
18
|
+
function isGrokImagineImageModel(model) {
|
|
19
|
+
return model.startsWith("grok-imagine-image");
|
|
20
|
+
}
|
|
21
|
+
function parseGrokImagineSize(size) {
|
|
22
|
+
const match = size.match(/^([\d.]+:[\d.]+|auto)(?:_(.+))?$/);
|
|
23
|
+
const [, aspectRatio, resolution] = match ?? [];
|
|
24
|
+
if (aspectRatio === void 0) return void 0;
|
|
25
|
+
return { aspectRatio, ...resolution !== void 0 && { resolution } };
|
|
26
|
+
}
|
|
1
27
|
function validateImageSize(model, size) {
|
|
2
28
|
if (!size) return;
|
|
29
|
+
if (isGrokImagineImageModel(model)) {
|
|
30
|
+
const parsed = parseGrokImagineSize(size);
|
|
31
|
+
if (!parsed || !GROK_IMAGINE_ASPECT_RATIOS.includes(parsed.aspectRatio) || parsed.resolution !== void 0 && !GROK_IMAGINE_RESOLUTIONS.includes(parsed.resolution)) {
|
|
32
|
+
throw new Error(
|
|
33
|
+
`Size "${size}" is not supported by model "${model}". Expected an aspect ratio (${GROK_IMAGINE_ASPECT_RATIOS.join(", ")}) optionally suffixed with a resolution ("16:9_2k"; resolutions: ${GROK_IMAGINE_RESOLUTIONS.join(", ")}).`
|
|
34
|
+
);
|
|
35
|
+
}
|
|
36
|
+
return;
|
|
37
|
+
}
|
|
3
38
|
const validSizes = {
|
|
4
39
|
"grok-2-image-1212": ["1024x1024", "1536x1024", "1024x1536"]
|
|
5
40
|
};
|
|
@@ -32,6 +67,8 @@ const validatePrompt = (options) => {
|
|
|
32
67
|
}
|
|
33
68
|
};
|
|
34
69
|
export {
|
|
70
|
+
isGrokImagineImageModel,
|
|
71
|
+
parseGrokImagineSize,
|
|
35
72
|
validateImageSize,
|
|
36
73
|
validateNumberOfImages,
|
|
37
74
|
validatePrompt
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"image-provider-options.js","sources":["../../../src/image/image-provider-options.ts"],"sourcesContent":["/**\n * Grok Image Generation Provider Options\n *\n * These are provider-specific options for Grok image generation.\n * Grok uses the grok-2-image-1212 model for image generation.\n */\n\n/**\n * Supported sizes for grok-2-image-1212 model\n */\nexport type GrokImageSize = '1024x1024' | '1536x1024' | '1024x1536'\n\n/**\n * Base provider options for Grok image models\n */\nexport interface GrokImageBaseProviderOptions {\n /**\n * A unique identifier representing your end-user.\n * Can help xAI to monitor and detect abuse.\n */\n user?: string\n}\n\n/**\n * Provider options for grok-2-image-1212 model\n */\nexport interface GrokImageProviderOptions extends GrokImageBaseProviderOptions {\n /**\n * The quality of the image.\n * @default 'standard'\n */\n quality?: 'standard' | 'hd'\n\n /**\n * The format in which generated images are returned.\n * URLs are only valid for 60 minutes after generation.\n * @default 'url'\n */\n response_format?: 'url' | 'b64_json'\n}\n\n/**\n * Type-only map from model name to its specific provider options.\n */\nexport type GrokImageModelProviderOptionsByName = {\n 'grok-2-image-1212': GrokImageProviderOptions\n}\n\n/**\n * Type-only map from model name to its supported sizes.\n */\nexport type GrokImageModelSizeByName = {\n 'grok-2-image-1212': GrokImageSize\n}\n\n/**\n * Internal options interface for validation\n */\ninterface ImageValidationOptions {\n prompt: string\n model: string\n}\n\n/**\n * Validates that the provided size is supported by the model.\n * Throws a descriptive error if the size is not supported.\n */\nexport function validateImageSize(\n model: string,\n size: string | undefined,\n): void {\n if (!size) return\n\n const validSizes: Record<string, Array<string>> = {\n 'grok-2-image-1212': ['1024x1024', '1536x1024', '1024x1536'],\n }\n\n const modelSizes = validSizes[model]\n if (!modelSizes) {\n throw new Error(`Unknown image model: ${model}`)\n }\n\n if (!modelSizes.includes(size)) {\n throw new Error(\n `Size \"${size}\" is not supported by model \"${model}\". ` +\n `Supported sizes: ${modelSizes.join(', ')}`,\n )\n }\n}\n\n/**\n * Validates that the number of images is within bounds for the model.\n */\nexport function validateNumberOfImages(\n _model: string,\n numberOfImages: number | undefined,\n): void {\n if (numberOfImages === undefined) return\n\n // grok-2-image-1212 supports 1-10 images per request\n if (numberOfImages < 1 || numberOfImages > 10) {\n throw new Error(\n `Number of images must be between 1 and 10. Requested: ${numberOfImages}`,\n )\n }\n}\n\nexport const validatePrompt = (options: ImageValidationOptions) => {\n if (options.prompt.length === 0) {\n throw new Error('Prompt cannot be empty.')\n }\n // Grok image model supports up to 4000 characters\n if (options.prompt.length > 4000) {\n throw new Error(\n 'For grok-2-image-1212, prompt length must be less than or equal to 4000 characters.',\n )\n }\n}\n"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"image-provider-options.js","sources":["../../../src/image/image-provider-options.ts"],"sourcesContent":["/**\n * Grok Image Generation Provider Options\n *\n * These are provider-specific options for Grok image generation.\n * Grok uses the grok-2-image-1212 model for image generation.\n */\n\n/**\n * Supported sizes for grok-2-image-1212 model\n */\nexport type GrokImageSize = '1024x1024' | '1536x1024' | '1024x1536'\n\n/**\n * Aspect ratios accepted by the grok-imagine image models.\n */\nexport type GrokImagineAspectRatio =\n | '1:1'\n | '3:4'\n | '4:3'\n | '9:16'\n | '16:9'\n | '2:3'\n | '3:2'\n | '9:19.5'\n | '19.5:9'\n | '9:20'\n | '20:9'\n | '1:2'\n | '2:1'\n | 'auto'\n\n/**\n * Resolution tiers for the grok-imagine image models.\n */\nexport type GrokImagineResolution = '1k' | '2k'\n\n/**\n * Size strings for grok-imagine image models. The Imagine API is\n * aspect-ratio based rather than pixel-size based; like Gemini's native\n * image models, the generic `size` option uses an\n * `aspectRatio_resolution` template (\"16:9_2k\") — the resolution suffix is\n * optional (\"16:9\" uses the API default of 1k).\n */\nexport type GrokImagineImageSize =\n | GrokImagineAspectRatio\n | `${GrokImagineAspectRatio}_${GrokImagineResolution}`\n\nconst GROK_IMAGINE_ASPECT_RATIOS: ReadonlyArray<string> = [\n '1:1',\n '3:4',\n '4:3',\n '9:16',\n '16:9',\n '2:3',\n '3:2',\n '9:19.5',\n '19.5:9',\n '9:20',\n '20:9',\n '1:2',\n '2:1',\n 'auto',\n]\n\nconst GROK_IMAGINE_RESOLUTIONS: ReadonlyArray<string> = ['1k', '2k']\n\n/**\n * Models served by xAI's Imagine API. They are aspect-ratio sized and\n * support image-conditioned generation via `/v1/images/edits`; the legacy\n * grok-2-image-1212 model is pixel-sized and text-to-image only.\n */\nexport function isGrokImagineImageModel(model: string): boolean {\n return model.startsWith('grok-imagine-image')\n}\n\n/**\n * Parses a grok-imagine size string into its components.\n * Format: \"aspectRatio\" or \"aspectRatio_resolution\",\n * e.g. \"16:9_2k\" → { aspectRatio: \"16:9\", resolution: \"2k\" }.\n * Returns undefined when the string doesn't match the template.\n */\nexport function parseGrokImagineSize(\n size: string,\n): { aspectRatio: string; resolution?: string } | undefined {\n const match = size.match(/^([\\d.]+:[\\d.]+|auto)(?:_(.+))?$/)\n const [, aspectRatio, resolution] = match ?? []\n if (aspectRatio === undefined) return undefined\n return { aspectRatio, ...(resolution !== undefined && { resolution }) }\n}\n\n/**\n * Base provider options for Grok image models\n */\nexport interface GrokImageBaseProviderOptions {\n /**\n * A unique identifier representing your end-user.\n * Can help xAI to monitor and detect abuse.\n */\n user?: string\n}\n\n/**\n * Provider options for grok-2-image-1212 model\n */\nexport interface GrokImageProviderOptions extends GrokImageBaseProviderOptions {\n /**\n * The quality of the image.\n * @default 'standard'\n */\n quality?: 'standard' | 'hd'\n\n /**\n * The format in which generated images are returned.\n * URLs are only valid for 60 minutes after generation.\n * @default 'url'\n */\n response_format?: 'url' | 'b64_json'\n}\n\n/**\n * Provider options for the grok-imagine image models (generation and\n * image-conditioned editing via xAI's Imagine API).\n */\nexport interface GrokImagineImageProviderOptions extends GrokImageBaseProviderOptions {\n /**\n * The format in which generated images are returned.\n * @default 'url'\n */\n response_format?: 'url' | 'b64_json'\n\n /**\n * Output resolution.\n * @default '1k'\n */\n resolution?: '1k' | '2k'\n\n /**\n * Processing tier for the request.\n * @default 'default'\n */\n service_tier?: 'default' | 'priority'\n}\n\n/**\n * Type-only map from model name to its specific provider options.\n */\nexport type GrokImageModelProviderOptionsByName = {\n 'grok-2-image-1212': GrokImageProviderOptions\n 'grok-imagine-image': GrokImagineImageProviderOptions\n 'grok-imagine-image-quality': GrokImagineImageProviderOptions\n}\n\n/**\n * Type-only map from model name to its supported sizes.\n */\nexport type GrokImageModelSizeByName = {\n 'grok-2-image-1212': GrokImageSize\n 'grok-imagine-image': GrokImagineImageSize\n 'grok-imagine-image-quality': GrokImagineImageSize\n}\n\n/**\n * Per-model prompt input modalities. Imagine API models accept image parts\n * in the prompt (routed to `/v1/images/edits`, up to 3 images, addressed by\n * xAI in request order); grok-2-image is text-to-image only.\n */\nexport type GrokImageModelInputModalitiesByName = {\n 'grok-2-image-1212': readonly []\n 'grok-imagine-image': readonly ['image']\n 'grok-imagine-image-quality': readonly ['image']\n}\n\n/**\n * Internal options interface for validation\n */\ninterface ImageValidationOptions {\n prompt: string\n model: string\n}\n\n/**\n * Validates that the provided size is supported by the model.\n * Throws a descriptive error if the size is not supported.\n */\nexport function validateImageSize(\n model: string,\n size: string | undefined,\n): void {\n if (!size) return\n\n if (isGrokImagineImageModel(model)) {\n const parsed = parseGrokImagineSize(size)\n if (\n !parsed ||\n !GROK_IMAGINE_ASPECT_RATIOS.includes(parsed.aspectRatio) ||\n (parsed.resolution !== undefined &&\n !GROK_IMAGINE_RESOLUTIONS.includes(parsed.resolution))\n ) {\n throw new Error(\n `Size \"${size}\" is not supported by model \"${model}\". ` +\n `Expected an aspect ratio (${GROK_IMAGINE_ASPECT_RATIOS.join(', ')}) ` +\n `optionally suffixed with a resolution (\"16:9_2k\"; resolutions: ${GROK_IMAGINE_RESOLUTIONS.join(', ')}).`,\n )\n }\n return\n }\n\n const validSizes: Record<string, Array<string>> = {\n 'grok-2-image-1212': ['1024x1024', '1536x1024', '1024x1536'],\n }\n\n const modelSizes = validSizes[model]\n if (!modelSizes) {\n throw new Error(`Unknown image model: ${model}`)\n }\n\n if (!modelSizes.includes(size)) {\n throw new Error(\n `Size \"${size}\" is not supported by model \"${model}\". ` +\n `Supported sizes: ${modelSizes.join(', ')}`,\n )\n }\n}\n\n/**\n * Validates that the number of images is within bounds for the model.\n */\nexport function validateNumberOfImages(\n _model: string,\n numberOfImages: number | undefined,\n): void {\n if (numberOfImages === undefined) return\n\n // grok-2-image-1212 supports 1-10 images per request\n if (numberOfImages < 1 || numberOfImages > 10) {\n throw new Error(\n `Number of images must be between 1 and 10. Requested: ${numberOfImages}`,\n )\n }\n}\n\nexport const validatePrompt = (options: ImageValidationOptions) => {\n if (options.prompt.length === 0) {\n throw new Error('Prompt cannot be empty.')\n }\n // Grok image model supports up to 4000 characters\n if (options.prompt.length > 4000) {\n throw new Error(\n 'For grok-2-image-1212, prompt length must be less than or equal to 4000 characters.',\n )\n }\n}\n"],"names":[],"mappings":"AA+CA,MAAM,6BAAoD;AAAA,EACxD;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAEA,MAAM,2BAAkD,CAAC,MAAM,IAAI;AAO5D,SAAS,wBAAwB,OAAwB;AAC9D,SAAO,MAAM,WAAW,oBAAoB;AAC9C;AAQO,SAAS,qBACd,MAC0D;AAC1D,QAAM,QAAQ,KAAK,MAAM,kCAAkC;AAC3D,QAAM,GAAG,aAAa,UAAU,IAAI,SAAS,CAAA;AAC7C,MAAI,gBAAgB,OAAW,QAAO;AACtC,SAAO,EAAE,aAAa,GAAI,eAAe,UAAa,EAAE,aAAW;AACrE;AAgGO,SAAS,kBACd,OACA,MACM;AACN,MAAI,CAAC,KAAM;AAEX,MAAI,wBAAwB,KAAK,GAAG;AAClC,UAAM,SAAS,qBAAqB,IAAI;AACxC,QACE,CAAC,UACD,CAAC,2BAA2B,SAAS,OAAO,WAAW,KACtD,OAAO,eAAe,UACrB,CAAC,yBAAyB,SAAS,OAAO,UAAU,GACtD;AACA,YAAM,IAAI;AAAA,QACR,SAAS,IAAI,gCAAgC,KAAK,gCACnB,2BAA2B,KAAK,IAAI,CAAC,oEACA,yBAAyB,KAAK,IAAI,CAAC;AAAA,MAAA;AAAA,IAE3G;AACA;AAAA,EACF;AAEA,QAAM,aAA4C;AAAA,IAChD,qBAAqB,CAAC,aAAa,aAAa,WAAW;AAAA,EAAA;AAG7D,QAAM,aAAa,WAAW,KAAK;AACnC,MAAI,CAAC,YAAY;AACf,UAAM,IAAI,MAAM,wBAAwB,KAAK,EAAE;AAAA,EACjD;AAEA,MAAI,CAAC,WAAW,SAAS,IAAI,GAAG;AAC9B,UAAM,IAAI;AAAA,MACR,SAAS,IAAI,gCAAgC,KAAK,uBAC5B,WAAW,KAAK,IAAI,CAAC;AAAA,IAAA;AAAA,EAE/C;AACF;AAKO,SAAS,uBACd,QACA,gBACM;AACN,MAAI,mBAAmB,OAAW;AAGlC,MAAI,iBAAiB,KAAK,iBAAiB,IAAI;AAC7C,UAAM,IAAI;AAAA,MACR,yDAAyD,cAAc;AAAA,IAAA;AAAA,EAE3E;AACF;AAEO,MAAM,iBAAiB,CAAC,YAAoC;AACjE,MAAI,QAAQ,OAAO,WAAW,GAAG;AAC/B,UAAM,IAAI,MAAM,yBAAyB;AAAA,EAC3C;AAEA,MAAI,QAAQ,OAAO,SAAS,KAAM;AAChC,UAAM,IAAI;AAAA,MACR;AAAA,IAAA;AAAA,EAEJ;AACF;"}
|
package/dist/esm/model-meta.d.ts
CHANGED
|
@@ -265,7 +265,7 @@ export declare const GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS: Set<string>;
|
|
|
265
265
|
/**
|
|
266
266
|
* Grok Image Generation Models
|
|
267
267
|
*/
|
|
268
|
-
export declare const GROK_IMAGE_MODELS: readonly ["grok-2-image-1212"];
|
|
268
|
+
export declare const GROK_IMAGE_MODELS: readonly ["grok-2-image-1212", "grok-imagine-image", "grok-imagine-image-quality"];
|
|
269
269
|
export declare const GROK_TTS_MODELS: readonly ["grok-tts"];
|
|
270
270
|
export declare const GROK_TRANSCRIPTION_MODELS: readonly ["grok-stt"];
|
|
271
271
|
export declare const GROK_REALTIME_MODELS: readonly ["grok-voice-fast-1.0", "grok-voice-think-fast-1.0"];
|
package/dist/esm/model-meta.js
CHANGED
|
@@ -28,6 +28,12 @@ const GROK_2_VISION = {
|
|
|
28
28
|
const GROK_2_IMAGE = {
|
|
29
29
|
name: "grok-2-image-1212"
|
|
30
30
|
};
|
|
31
|
+
const GROK_IMAGINE_IMAGE = {
|
|
32
|
+
name: "grok-imagine-image"
|
|
33
|
+
};
|
|
34
|
+
const GROK_IMAGINE_IMAGE_QUALITY = {
|
|
35
|
+
name: "grok-imagine-image-quality"
|
|
36
|
+
};
|
|
31
37
|
const GROK_4_20 = {
|
|
32
38
|
name: "grok-4.20"
|
|
33
39
|
};
|
|
@@ -66,7 +72,11 @@ const GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS = /* @__PURE__ */ new Set([
|
|
|
66
72
|
GROK_4_20_MULTI_AGENT.name,
|
|
67
73
|
GROK_4_3.name
|
|
68
74
|
]);
|
|
69
|
-
const GROK_IMAGE_MODELS = [
|
|
75
|
+
const GROK_IMAGE_MODELS = [
|
|
76
|
+
GROK_2_IMAGE.name,
|
|
77
|
+
GROK_IMAGINE_IMAGE.name,
|
|
78
|
+
GROK_IMAGINE_IMAGE_QUALITY.name
|
|
79
|
+
];
|
|
70
80
|
const GROK_TTS = {
|
|
71
81
|
name: "grok-tts"
|
|
72
82
|
};
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"model-meta.js","sources":["../../src/model-meta.ts"],"sourcesContent":["/**\n * Model metadata interface for documentation and type inference\n */\ninterface ModelMeta {\n name: string\n supports: {\n input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>\n output: Array<'text' | 'image' | 'audio' | 'video'>\n capabilities?: Array<'reasoning' | 'tool_calling' | 'structured_outputs'>\n tools?: ReadonlyArray<never>\n }\n max_input_tokens?: number\n max_output_tokens?: number\n context_window?: number\n knowledge_cutoff?: string\n pricing?: {\n input: {\n normal: number\n cached?: number\n }\n output: {\n normal: number\n }\n }\n}\n\nconst GROK_4_1_FAST_REASONING = {\n name: 'grok-4-1-fast-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_1_FAST_NON_REASONING = {\n name: 'grok-4-1-fast-non-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_CODE_FAST_1 = {\n name: 'grok-code-fast-1',\n context_window: 256_000,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.02,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_FAST_REASONING = {\n name: 'grok-4-fast-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_FAST_NON_REASONING = {\n name: 'grok-4-fast-non-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4 = {\n name: 'grok-4',\n context_window: 256_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 3,\n cached: 0.75,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_3_MINI = {\n name: 'grok-3-mini',\n context_window: 131_072,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.3,\n cached: 0.075,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_3 = {\n name: 'grok-3',\n context_window: 131_072,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 3,\n cached: 0.75,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_2_VISION = {\n name: 'grok-2-vision-1212',\n context_window: 32_768,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n },\n output: {\n normal: 10,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_2_IMAGE = {\n name: 'grok-2-image-1212',\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0.07,\n },\n output: {\n normal: 0.07,\n },\n },\n} as const satisfies ModelMeta\n\n/**\n * Grok Chat Models\n * Based on xAI's available models as of 2025\n */\nconst GROK_4_20 = {\n name: 'grok-4.20',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image', 'document'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n cached: 0.2,\n },\n output: {\n normal: 6,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_20_MULTI_AGENT = {\n name: 'grok-4.20-multi-agent',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image', 'document'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n cached: 0.2,\n },\n output: {\n normal: 6,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_3 = {\n name: 'grok-4.3',\n context_window: 1_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [],\n },\n pricing: {\n input: {\n normal: 1.25,\n cached: 0.2,\n },\n output: {\n normal: 2.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_BUILD_0_1 = {\n name: 'grok-build-0.1',\n context_window: 256_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [],\n },\n pricing: {\n input: {\n normal: 1,\n cached: 0.2,\n },\n output: {\n normal: 2,\n },\n },\n} as const satisfies ModelMeta\n\nexport const GROK_CHAT_MODELS = [\n GROK_4_1_FAST_REASONING.name,\n GROK_4_1_FAST_NON_REASONING.name,\n GROK_CODE_FAST_1.name,\n GROK_4_FAST_REASONING.name,\n GROK_4_FAST_NON_REASONING.name,\n GROK_4.name,\n GROK_3.name,\n GROK_3_MINI.name,\n GROK_2_VISION.name,\n\n GROK_4_20.name,\n GROK_4_20_MULTI_AGENT.name,\n\n GROK_4_3.name,\n\n GROK_BUILD_0_1.name,\n] as const\n\n/**\n * Grok models that support combining `tools` + `response_format: json_schema`\n * in a single streaming Chat Completions request (per issue #605). xAI\n * docs gate this to the Grok 4 family — Grok 2 / 3 reject the\n * combination. Grok 2 image generation is not a chat model, omitted.\n *\n * Note: Grok streams tool-call arguments atomically (not token-streamed)\n * per the issue's source matrix; partial-JSON tool-arg parsing should be\n * skipped for Grok specifically. That's a separate adapter concern from\n * this set — the set only gates whether the engine takes the native\n * combined path vs the legacy finalization path.\n */\nexport const GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS = new Set<string>([\n GROK_4_1_FAST_REASONING.name,\n GROK_4_1_FAST_NON_REASONING.name,\n GROK_CODE_FAST_1.name,\n GROK_4_FAST_REASONING.name,\n GROK_4_FAST_NON_REASONING.name,\n GROK_4.name,\n GROK_4_20.name,\n GROK_4_20_MULTI_AGENT.name,\n GROK_4_3.name,\n])\n\n/**\n * Grok Image Generation Models\n */\nexport const GROK_IMAGE_MODELS = [GROK_2_IMAGE.name] as const\n\n// xAI's `/v1/tts` endpoint is endpoint-addressed and does not take a `model`\n// parameter. This synthetic identifier satisfies the SDK's `TTSOptions.model`\n// contract and provides a stable value for logging and fixture matching.\nconst GROK_TTS = {\n name: 'grok-tts',\n supports: {\n input: ['text'],\n output: ['audio'],\n },\n} as const satisfies ModelMeta\n\n// xAI's `/v1/stt` endpoint is endpoint-addressed and does not take a `model`\n// parameter. This synthetic identifier satisfies the SDK's\n// `TranscriptionOptions.model` contract.\nconst GROK_STT = {\n name: 'grok-stt',\n supports: {\n input: ['audio'],\n output: ['text'],\n },\n} as const satisfies ModelMeta\n\nconst GROK_VOICE_FAST_1 = {\n name: 'grok-voice-fast-1.0',\n supports: {\n input: ['audio', 'text'],\n output: ['audio', 'text'],\n capabilities: ['tool_calling'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta\n\nconst GROK_VOICE_THINK_FAST_1 = {\n name: 'grok-voice-think-fast-1.0',\n supports: {\n input: ['audio', 'text'],\n output: ['audio', 'text'],\n capabilities: ['reasoning', 'tool_calling'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta\n\nexport const GROK_TTS_MODELS = [GROK_TTS.name] as const\n\nexport const GROK_TRANSCRIPTION_MODELS = [GROK_STT.name] as const\n\nexport const GROK_REALTIME_MODELS = [\n GROK_VOICE_FAST_1.name,\n GROK_VOICE_THINK_FAST_1.name,\n] as const\n\nexport type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]\nexport type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]\nexport type GrokTTSModel = (typeof GROK_TTS_MODELS)[number]\nexport type GrokTranscriptionModel = (typeof GROK_TRANSCRIPTION_MODELS)[number]\nexport type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]\n\n/**\n * Type-only map from Grok chat model name to its supported input modalities.\n * Used for type inference when constructing multimodal messages.\n */\nexport type GrokModelInputModalitiesByName = {\n [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.input\n [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.input\n [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.input\n [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.input\n [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.input\n [GROK_4.name]: typeof GROK_4.supports.input\n [GROK_3.name]: typeof GROK_3.supports.input\n [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.input\n [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.input\n [GROK_4_20.name]: typeof GROK_4_20.supports.input\n [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.input\n [GROK_4_3.name]: typeof GROK_4_3.supports.input\n [GROK_BUILD_0_1.name]: typeof GROK_BUILD_0_1.supports.input\n}\n\n/**\n * Type-only map from Grok chat model name to its provider options type.\n * Since Grok uses OpenAI-compatible API, we reuse OpenAI provider options.\n */\nexport type GrokChatModelProviderOptionsByName = {\n [K in (typeof GROK_CHAT_MODELS)[number]]: GrokProviderOptions\n}\n\n/**\n * Type-only map from Grok chat model name to its supported provider tools.\n * Grok exposes no provider-specific tool factories, so every model gets an\n * empty tuple. This ensures that passing an Anthropic/OpenAI ProviderTool to\n * a Grok adapter produces a compile-time type error.\n */\nexport type GrokChatModelToolCapabilitiesByName = {\n [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.tools\n [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.tools\n [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.tools\n [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.tools\n [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.tools\n [GROK_4.name]: typeof GROK_4.supports.tools\n [GROK_3.name]: typeof GROK_3.supports.tools\n [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.tools\n [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.tools\n [GROK_4_20.name]: typeof GROK_4_20.supports.tools\n [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.tools\n}\n\n/**\n * Grok-specific provider options\n * Based on OpenAI-compatible API options\n */\nexport interface GrokProviderOptions {\n /** Temperature for response generation (0-2) */\n temperature?: number\n /** Maximum tokens in the response */\n max_tokens?: number\n /** Top-p sampling parameter */\n top_p?: number\n /** Frequency penalty (-2.0 to 2.0) */\n frequency_penalty?: number\n /** Presence penalty (-2.0 to 2.0) */\n presence_penalty?: number\n /** Stop sequences */\n stop?: string | Array<string>\n /** A unique identifier representing your end-user */\n user?: string\n}\n\n// ===========================\n// Type Resolution Helpers\n// ===========================\n\n/**\n * Resolve provider options for a specific model.\n * If the model has explicit options in the map, use those; otherwise use base options.\n */\nexport type ResolveProviderOptions<TModel extends string> =\n TModel extends keyof GrokChatModelProviderOptionsByName\n ? GrokChatModelProviderOptionsByName[TModel]\n : GrokProviderOptions\n\n/**\n * Resolve input modalities for a specific model.\n * If the model has explicit modalities in the map, use those; otherwise use text only.\n */\nexport type ResolveInputModalities<TModel extends string> =\n TModel extends keyof GrokModelInputModalitiesByName\n ? GrokModelInputModalitiesByName[TModel]\n : readonly ['text']\n"],"names":[],"mappings":"AA0BA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAiBR;AAEA,MAAM,8BAA8B;AAAA,EAClC,MAAM;AAiBR;AAEA,MAAM,mBAAmB;AAAA,EACvB,MAAM;AAiBR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAiBR;AAEA,MAAM,4BAA4B;AAAA,EAChC,MAAM;AAiBR;AAEA,MAAM,SAAS;AAAA,EACb,MAAM;AAiBR;AAEA,MAAM,cAAc;AAAA,EAClB,MAAM;AAiBR;AAEA,MAAM,SAAS;AAAA,EACb,MAAM;AAiBR;AAEA,MAAM,gBAAgB;AAAA,EACpB,MAAM;AAgBR;AAEA,MAAM,eAAe;AAAA,EACnB,MAAM;AAaR;AAMA,MAAM,YAAY;AAAA,EAChB,MAAM;AAiBR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAiBR;AAEA,MAAM,WAAW;AAAA,EACf,MAAM;AAiBR;AAEA,MAAM,iBAAiB;AAAA,EACrB,MAAM;AAiBR;AAEO,MAAM,mBAAmB;AAAA,EAC9B,wBAAwB;AAAA,EACxB,4BAA4B;AAAA,EAC5B,iBAAiB;AAAA,EACjB,sBAAsB;AAAA,EACtB,0BAA0B;AAAA,EAC1B,OAAO;AAAA,EACP,OAAO;AAAA,EACP,YAAY;AAAA,EACZ,cAAc;AAAA,EAEd,UAAU;AAAA,EACV,sBAAsB;AAAA,EAEtB,SAAS;AAAA,EAET,eAAe;AACjB;AAcO,MAAM,4DAA4C,IAAY;AAAA,EACnE,wBAAwB;AAAA,EACxB,4BAA4B;AAAA,EAC5B,iBAAiB;AAAA,EACjB,sBAAsB;AAAA,EACtB,0BAA0B;AAAA,EAC1B,OAAO;AAAA,EACP,UAAU;AAAA,EACV,sBAAsB;AAAA,EACtB,SAAS;AACX,CAAC;AAKM,MAAM,oBAAoB,CAAC,aAAa,IAAI;AAKnD,MAAM,WAAW;AAAA,EACf,MAAM;AAKR;AAKA,MAAM,WAAW;AAAA,EACf,MAAM;AAKR;AAEA,MAAM,oBAAoB;AAAA,EACxB,MAAM;AAOR;AAEA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAOR;AAEO,MAAM,kBAAkB,CAAC,SAAS,IAAI;AAEtC,MAAM,4BAA4B,CAAC,SAAS,IAAI;AAEhD,MAAM,uBAAuB;AAAA,EAClC,kBAAkB;AAAA,EAClB,wBAAwB;AAC1B;"}
|
|
1
|
+
{"version":3,"file":"model-meta.js","sources":["../../src/model-meta.ts"],"sourcesContent":["/**\n * Model metadata interface for documentation and type inference\n */\ninterface ModelMeta {\n name: string\n supports: {\n input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>\n output: Array<'text' | 'image' | 'audio' | 'video'>\n capabilities?: Array<'reasoning' | 'tool_calling' | 'structured_outputs'>\n tools?: ReadonlyArray<never>\n }\n max_input_tokens?: number\n max_output_tokens?: number\n context_window?: number\n knowledge_cutoff?: string\n pricing?: {\n input: {\n normal: number\n cached?: number\n }\n output: {\n normal: number\n }\n }\n}\n\nconst GROK_4_1_FAST_REASONING = {\n name: 'grok-4-1-fast-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_1_FAST_NON_REASONING = {\n name: 'grok-4-1-fast-non-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_CODE_FAST_1 = {\n name: 'grok-code-fast-1',\n context_window: 256_000,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.02,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_FAST_REASONING = {\n name: 'grok-4-fast-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_FAST_NON_REASONING = {\n name: 'grok-4-fast-non-reasoning',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.2,\n cached: 0.05,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4 = {\n name: 'grok-4',\n context_window: 256_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 3,\n cached: 0.75,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_3_MINI = {\n name: 'grok-3-mini',\n context_window: 131_072,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 0.3,\n cached: 0.075,\n },\n output: {\n normal: 0.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_3 = {\n name: 'grok-3',\n context_window: 131_072,\n supports: {\n input: ['text'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 3,\n cached: 0.75,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_2_VISION = {\n name: 'grok-2-vision-1212',\n context_window: 32_768,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n },\n output: {\n normal: 10,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_2_IMAGE = {\n name: 'grok-2-image-1212',\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0.07,\n },\n output: {\n normal: 0.07,\n },\n },\n} as const satisfies ModelMeta\n\n// Imagine API image models. Pricing is per generated image (output only).\nconst GROK_IMAGINE_IMAGE = {\n name: 'grok-imagine-image',\n supports: {\n input: ['text', 'image'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.02,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_IMAGINE_IMAGE_QUALITY = {\n name: 'grok-imagine-image-quality',\n supports: {\n input: ['text', 'image'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.05,\n },\n },\n} as const satisfies ModelMeta\n\n/**\n * Grok Chat Models\n * Based on xAI's available models as of 2025\n */\nconst GROK_4_20 = {\n name: 'grok-4.20',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image', 'document'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n cached: 0.2,\n },\n output: {\n normal: 6,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_20_MULTI_AGENT = {\n name: 'grok-4.20-multi-agent',\n context_window: 2_000_000,\n supports: {\n input: ['text', 'image', 'document'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [] as const,\n },\n pricing: {\n input: {\n normal: 2,\n cached: 0.2,\n },\n output: {\n normal: 6,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_4_3 = {\n name: 'grok-4.3',\n context_window: 1_000_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [],\n },\n pricing: {\n input: {\n normal: 1.25,\n cached: 0.2,\n },\n output: {\n normal: 2.5,\n },\n },\n} as const satisfies ModelMeta\n\nconst GROK_BUILD_0_1 = {\n name: 'grok-build-0.1',\n context_window: 256_000,\n supports: {\n input: ['text', 'image'],\n output: ['text'],\n capabilities: ['reasoning', 'structured_outputs', 'tool_calling'],\n tools: [],\n },\n pricing: {\n input: {\n normal: 1,\n cached: 0.2,\n },\n output: {\n normal: 2,\n },\n },\n} as const satisfies ModelMeta\n\nexport const GROK_CHAT_MODELS = [\n GROK_4_1_FAST_REASONING.name,\n GROK_4_1_FAST_NON_REASONING.name,\n GROK_CODE_FAST_1.name,\n GROK_4_FAST_REASONING.name,\n GROK_4_FAST_NON_REASONING.name,\n GROK_4.name,\n GROK_3.name,\n GROK_3_MINI.name,\n GROK_2_VISION.name,\n\n GROK_4_20.name,\n GROK_4_20_MULTI_AGENT.name,\n\n GROK_4_3.name,\n\n GROK_BUILD_0_1.name,\n] as const\n\n/**\n * Grok models that support combining `tools` + `response_format: json_schema`\n * in a single streaming Chat Completions request (per issue #605). xAI\n * docs gate this to the Grok 4 family — Grok 2 / 3 reject the\n * combination. Grok 2 image generation is not a chat model, omitted.\n *\n * Note: Grok streams tool-call arguments atomically (not token-streamed)\n * per the issue's source matrix; partial-JSON tool-arg parsing should be\n * skipped for Grok specifically. That's a separate adapter concern from\n * this set — the set only gates whether the engine takes the native\n * combined path vs the legacy finalization path.\n */\nexport const GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS = new Set<string>([\n GROK_4_1_FAST_REASONING.name,\n GROK_4_1_FAST_NON_REASONING.name,\n GROK_CODE_FAST_1.name,\n GROK_4_FAST_REASONING.name,\n GROK_4_FAST_NON_REASONING.name,\n GROK_4.name,\n GROK_4_20.name,\n GROK_4_20_MULTI_AGENT.name,\n GROK_4_3.name,\n])\n\n/**\n * Grok Image Generation Models\n */\nexport const GROK_IMAGE_MODELS = [\n GROK_2_IMAGE.name,\n GROK_IMAGINE_IMAGE.name,\n GROK_IMAGINE_IMAGE_QUALITY.name,\n] as const\n\n// xAI's `/v1/tts` endpoint is endpoint-addressed and does not take a `model`\n// parameter. This synthetic identifier satisfies the SDK's `TTSOptions.model`\n// contract and provides a stable value for logging and fixture matching.\nconst GROK_TTS = {\n name: 'grok-tts',\n supports: {\n input: ['text'],\n output: ['audio'],\n },\n} as const satisfies ModelMeta\n\n// xAI's `/v1/stt` endpoint is endpoint-addressed and does not take a `model`\n// parameter. This synthetic identifier satisfies the SDK's\n// `TranscriptionOptions.model` contract.\nconst GROK_STT = {\n name: 'grok-stt',\n supports: {\n input: ['audio'],\n output: ['text'],\n },\n} as const satisfies ModelMeta\n\nconst GROK_VOICE_FAST_1 = {\n name: 'grok-voice-fast-1.0',\n supports: {\n input: ['audio', 'text'],\n output: ['audio', 'text'],\n capabilities: ['tool_calling'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta\n\nconst GROK_VOICE_THINK_FAST_1 = {\n name: 'grok-voice-think-fast-1.0',\n supports: {\n input: ['audio', 'text'],\n output: ['audio', 'text'],\n capabilities: ['reasoning', 'tool_calling'],\n tools: [] as const,\n },\n} as const satisfies ModelMeta\n\nexport const GROK_TTS_MODELS = [GROK_TTS.name] as const\n\nexport const GROK_TRANSCRIPTION_MODELS = [GROK_STT.name] as const\n\nexport const GROK_REALTIME_MODELS = [\n GROK_VOICE_FAST_1.name,\n GROK_VOICE_THINK_FAST_1.name,\n] as const\n\nexport type GrokChatModel = (typeof GROK_CHAT_MODELS)[number]\nexport type GrokImageModel = (typeof GROK_IMAGE_MODELS)[number]\nexport type GrokTTSModel = (typeof GROK_TTS_MODELS)[number]\nexport type GrokTranscriptionModel = (typeof GROK_TRANSCRIPTION_MODELS)[number]\nexport type GrokRealtimeModel = (typeof GROK_REALTIME_MODELS)[number]\n\n/**\n * Type-only map from Grok chat model name to its supported input modalities.\n * Used for type inference when constructing multimodal messages.\n */\nexport type GrokModelInputModalitiesByName = {\n [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.input\n [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.input\n [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.input\n [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.input\n [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.input\n [GROK_4.name]: typeof GROK_4.supports.input\n [GROK_3.name]: typeof GROK_3.supports.input\n [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.input\n [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.input\n [GROK_4_20.name]: typeof GROK_4_20.supports.input\n [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.input\n [GROK_4_3.name]: typeof GROK_4_3.supports.input\n [GROK_BUILD_0_1.name]: typeof GROK_BUILD_0_1.supports.input\n}\n\n/**\n * Type-only map from Grok chat model name to its provider options type.\n * Since Grok uses OpenAI-compatible API, we reuse OpenAI provider options.\n */\nexport type GrokChatModelProviderOptionsByName = {\n [K in (typeof GROK_CHAT_MODELS)[number]]: GrokProviderOptions\n}\n\n/**\n * Type-only map from Grok chat model name to its supported provider tools.\n * Grok exposes no provider-specific tool factories, so every model gets an\n * empty tuple. This ensures that passing an Anthropic/OpenAI ProviderTool to\n * a Grok adapter produces a compile-time type error.\n */\nexport type GrokChatModelToolCapabilitiesByName = {\n [GROK_4_1_FAST_REASONING.name]: typeof GROK_4_1_FAST_REASONING.supports.tools\n [GROK_4_1_FAST_NON_REASONING.name]: typeof GROK_4_1_FAST_NON_REASONING.supports.tools\n [GROK_CODE_FAST_1.name]: typeof GROK_CODE_FAST_1.supports.tools\n [GROK_4_FAST_REASONING.name]: typeof GROK_4_FAST_REASONING.supports.tools\n [GROK_4_FAST_NON_REASONING.name]: typeof GROK_4_FAST_NON_REASONING.supports.tools\n [GROK_4.name]: typeof GROK_4.supports.tools\n [GROK_3.name]: typeof GROK_3.supports.tools\n [GROK_3_MINI.name]: typeof GROK_3_MINI.supports.tools\n [GROK_2_VISION.name]: typeof GROK_2_VISION.supports.tools\n [GROK_4_20.name]: typeof GROK_4_20.supports.tools\n [GROK_4_20_MULTI_AGENT.name]: typeof GROK_4_20_MULTI_AGENT.supports.tools\n}\n\n/**\n * Grok-specific provider options\n * Based on OpenAI-compatible API options\n */\nexport interface GrokProviderOptions {\n /** Temperature for response generation (0-2) */\n temperature?: number\n /** Maximum tokens in the response */\n max_tokens?: number\n /** Top-p sampling parameter */\n top_p?: number\n /** Frequency penalty (-2.0 to 2.0) */\n frequency_penalty?: number\n /** Presence penalty (-2.0 to 2.0) */\n presence_penalty?: number\n /** Stop sequences */\n stop?: string | Array<string>\n /** A unique identifier representing your end-user */\n user?: string\n}\n\n// ===========================\n// Type Resolution Helpers\n// ===========================\n\n/**\n * Resolve provider options for a specific model.\n * If the model has explicit options in the map, use those; otherwise use base options.\n */\nexport type ResolveProviderOptions<TModel extends string> =\n TModel extends keyof GrokChatModelProviderOptionsByName\n ? GrokChatModelProviderOptionsByName[TModel]\n : GrokProviderOptions\n\n/**\n * Resolve input modalities for a specific model.\n * If the model has explicit modalities in the map, use those; otherwise use text only.\n */\nexport type ResolveInputModalities<TModel extends string> =\n TModel extends keyof GrokModelInputModalitiesByName\n ? GrokModelInputModalitiesByName[TModel]\n : readonly ['text']\n"],"names":[],"mappings":"AA0BA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAiBR;AAEA,MAAM,8BAA8B;AAAA,EAClC,MAAM;AAiBR;AAEA,MAAM,mBAAmB;AAAA,EACvB,MAAM;AAiBR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAiBR;AAEA,MAAM,4BAA4B;AAAA,EAChC,MAAM;AAiBR;AAEA,MAAM,SAAS;AAAA,EACb,MAAM;AAiBR;AAEA,MAAM,cAAc;AAAA,EAClB,MAAM;AAiBR;AAEA,MAAM,SAAS;AAAA,EACb,MAAM;AAiBR;AAEA,MAAM,gBAAgB;AAAA,EACpB,MAAM;AAgBR;AAEA,MAAM,eAAe;AAAA,EACnB,MAAM;AAaR;AAGA,MAAM,qBAAqB;AAAA,EACzB,MAAM;AAaR;AAEA,MAAM,6BAA6B;AAAA,EACjC,MAAM;AAaR;AAMA,MAAM,YAAY;AAAA,EAChB,MAAM;AAiBR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAiBR;AAEA,MAAM,WAAW;AAAA,EACf,MAAM;AAiBR;AAEA,MAAM,iBAAiB;AAAA,EACrB,MAAM;AAiBR;AAEO,MAAM,mBAAmB;AAAA,EAC9B,wBAAwB;AAAA,EACxB,4BAA4B;AAAA,EAC5B,iBAAiB;AAAA,EACjB,sBAAsB;AAAA,EACtB,0BAA0B;AAAA,EAC1B,OAAO;AAAA,EACP,OAAO;AAAA,EACP,YAAY;AAAA,EACZ,cAAc;AAAA,EAEd,UAAU;AAAA,EACV,sBAAsB;AAAA,EAEtB,SAAS;AAAA,EAET,eAAe;AACjB;AAcO,MAAM,4DAA4C,IAAY;AAAA,EACnE,wBAAwB;AAAA,EACxB,4BAA4B;AAAA,EAC5B,iBAAiB;AAAA,EACjB,sBAAsB;AAAA,EACtB,0BAA0B;AAAA,EAC1B,OAAO;AAAA,EACP,UAAU;AAAA,EACV,sBAAsB;AAAA,EACtB,SAAS;AACX,CAAC;AAKM,MAAM,oBAAoB;AAAA,EAC/B,aAAa;AAAA,EACb,mBAAmB;AAAA,EACnB,2BAA2B;AAC7B;AAKA,MAAM,WAAW;AAAA,EACf,MAAM;AAKR;AAKA,MAAM,WAAW;AAAA,EACf,MAAM;AAKR;AAEA,MAAM,oBAAoB;AAAA,EACxB,MAAM;AAOR;AAEA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAOR;AAEO,MAAM,kBAAkB,CAAC,SAAS,IAAI;AAEtC,MAAM,4BAA4B,CAAC,SAAS,IAAI;AAEhD,MAAM,uBAAuB;AAAA,EAClC,kBAAkB;AAAA,EAClB,wBAAwB;AAC1B;"}
|
package/package.json
CHANGED
|
@@ -1,14 +1,22 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai-grok",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.12.1",
|
|
4
4
|
"description": "xAI Grok adapter for TanStack AI chat, image generation, realtime, and structured outputs.",
|
|
5
|
-
"author": "",
|
|
5
|
+
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
7
|
+
"homepage": "https://tanstack.com/ai",
|
|
7
8
|
"repository": {
|
|
8
9
|
"type": "git",
|
|
9
10
|
"url": "git+https://github.com/TanStack/ai.git",
|
|
10
11
|
"directory": "packages/ai-grok"
|
|
11
12
|
},
|
|
13
|
+
"bugs": {
|
|
14
|
+
"url": "https://github.com/TanStack/ai/issues"
|
|
15
|
+
},
|
|
16
|
+
"funding": {
|
|
17
|
+
"type": "github",
|
|
18
|
+
"url": "https://github.com/sponsors/tannerlinsley"
|
|
19
|
+
},
|
|
12
20
|
"type": "module",
|
|
13
21
|
"module": "./dist/esm/index.js",
|
|
14
22
|
"types": "./dist/esm/index.d.ts",
|
|
@@ -42,18 +50,18 @@
|
|
|
42
50
|
],
|
|
43
51
|
"dependencies": {
|
|
44
52
|
"openai": "^6.41.0",
|
|
45
|
-
"@tanstack/ai-utils": "0.2.
|
|
46
|
-
"@tanstack/openai-base": "0.8.
|
|
53
|
+
"@tanstack/ai-utils": "0.2.2",
|
|
54
|
+
"@tanstack/openai-base": "0.8.6"
|
|
47
55
|
},
|
|
48
56
|
"devDependencies": {
|
|
49
57
|
"@vitest/coverage-v8": "4.0.14",
|
|
50
58
|
"vite": "^7.3.3",
|
|
51
|
-
"@tanstack/ai": "0.
|
|
52
|
-
"@tanstack/ai-client": "0.
|
|
59
|
+
"@tanstack/ai": "0.32.0",
|
|
60
|
+
"@tanstack/ai-client": "0.18.0"
|
|
53
61
|
},
|
|
54
62
|
"peerDependencies": {
|
|
55
63
|
"zod": "^4.0.0",
|
|
56
|
-
"@tanstack/ai": "^0.
|
|
64
|
+
"@tanstack/ai": "^0.32.0"
|
|
57
65
|
},
|
|
58
66
|
"scripts": {
|
|
59
67
|
"build": "vite build",
|
package/src/adapters/image.ts
CHANGED
|
@@ -1,10 +1,13 @@
|
|
|
1
1
|
import OpenAI from 'openai'
|
|
2
|
+
import { resolveMediaPrompt } from '@tanstack/ai'
|
|
2
3
|
import { BaseImageAdapter } from '@tanstack/ai/adapters'
|
|
3
4
|
import { toRunErrorPayload } from '@tanstack/ai/adapter-internals'
|
|
4
5
|
import { buildImagesUsage } from '@tanstack/openai-base'
|
|
5
6
|
import { generateId } from '@tanstack/ai-utils'
|
|
6
7
|
import { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'
|
|
7
8
|
import {
|
|
9
|
+
isGrokImagineImageModel,
|
|
10
|
+
parseGrokImagineSize,
|
|
8
11
|
validateImageSize,
|
|
9
12
|
validateNumberOfImages,
|
|
10
13
|
validatePrompt,
|
|
@@ -13,10 +16,14 @@ import type {
|
|
|
13
16
|
GeneratedImage,
|
|
14
17
|
ImageGenerationOptions,
|
|
15
18
|
ImageGenerationResult,
|
|
19
|
+
ImagePart,
|
|
20
|
+
MediaInputMetadata,
|
|
21
|
+
ResolvedMediaPrompt,
|
|
16
22
|
} from '@tanstack/ai'
|
|
17
23
|
import type OpenAI_SDK from 'openai'
|
|
18
24
|
import type { GrokImageModel } from '../model-meta'
|
|
19
25
|
import type {
|
|
26
|
+
GrokImageModelInputModalitiesByName,
|
|
20
27
|
GrokImageModelProviderOptionsByName,
|
|
21
28
|
GrokImageModelSizeByName,
|
|
22
29
|
GrokImageProviderOptions,
|
|
@@ -28,15 +35,58 @@ import type { GrokClientConfig } from '../utils'
|
|
|
28
35
|
*/
|
|
29
36
|
export interface GrokImageConfig extends GrokClientConfig {}
|
|
30
37
|
|
|
38
|
+
/** Maximum source images accepted by xAI's image edit endpoint. */
|
|
39
|
+
const MAX_EDIT_IMAGES = 3
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Maps the generic `size` option onto Imagine API parameters: the
|
|
43
|
+
* "aspectRatio_resolution" template ("16:9_2k") splits into `aspect_ratio`
|
|
44
|
+
* and optional `resolution` request fields.
|
|
45
|
+
*/
|
|
46
|
+
function imagineSizeParams(size: string | undefined): {
|
|
47
|
+
aspect_ratio?: string
|
|
48
|
+
resolution?: string
|
|
49
|
+
} {
|
|
50
|
+
if (!size) return {}
|
|
51
|
+
const parsed = parseGrokImagineSize(size)
|
|
52
|
+
if (!parsed) return {}
|
|
53
|
+
return {
|
|
54
|
+
aspect_ratio: parsed.aspectRatio,
|
|
55
|
+
...(parsed.resolution !== undefined && { resolution: parsed.resolution }),
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Convert a TanStack ImagePart to the URL string accepted by xAI's edit
|
|
61
|
+
* endpoint: public URLs pass through (fetched by xAI's servers), data
|
|
62
|
+
* sources become base64 data URIs.
|
|
63
|
+
*/
|
|
64
|
+
function imagePartToUrl(part: ImagePart<MediaInputMetadata>): string {
|
|
65
|
+
if (part.source.type === 'url') return part.source.value
|
|
66
|
+
return `data:${part.source.mimeType};base64,${part.source.value}`
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/** Response shape of xAI's `/v1/images/edits` endpoint. */
|
|
70
|
+
interface GrokImageEditResponse {
|
|
71
|
+
data?: Array<{
|
|
72
|
+
url?: string | null
|
|
73
|
+
b64_json?: string | null
|
|
74
|
+
mime_type?: string
|
|
75
|
+
}>
|
|
76
|
+
}
|
|
77
|
+
|
|
31
78
|
/**
|
|
32
79
|
* Grok Image Generation Adapter
|
|
33
80
|
*
|
|
34
81
|
* Tree-shakeable adapter for Grok image generation functionality.
|
|
35
|
-
* Supports grok-2-image-1212 model
|
|
82
|
+
* Supports the legacy grok-2-image-1212 model (text-to-image via the
|
|
83
|
+
* OpenAI-compat endpoint) and the grok-imagine image models, which also
|
|
84
|
+
* accept image prompt parts for image-conditioned generation via xAI's
|
|
85
|
+
* `/v1/images/edits` endpoint (up to 3 source images).
|
|
36
86
|
*
|
|
37
87
|
* Features:
|
|
38
88
|
* - Model-specific type-safe provider options
|
|
39
|
-
* - Size validation per model
|
|
89
|
+
* - Size / aspect-ratio validation per model
|
|
40
90
|
* - Number of images validation
|
|
41
91
|
*/
|
|
42
92
|
export class GrokImageAdapter<
|
|
@@ -45,36 +95,67 @@ export class GrokImageAdapter<
|
|
|
45
95
|
TModel,
|
|
46
96
|
GrokImageProviderOptions,
|
|
47
97
|
GrokImageModelProviderOptionsByName,
|
|
48
|
-
GrokImageModelSizeByName
|
|
98
|
+
GrokImageModelSizeByName,
|
|
99
|
+
GrokImageModelInputModalitiesByName
|
|
49
100
|
> {
|
|
50
101
|
override readonly kind = 'image' as const
|
|
51
102
|
readonly name = 'grok' as const
|
|
52
103
|
|
|
53
104
|
protected client: OpenAI
|
|
105
|
+
private readonly clientConfig: GrokImageConfig
|
|
54
106
|
|
|
55
107
|
constructor(config: GrokImageConfig, model: TModel) {
|
|
56
108
|
super(model, {})
|
|
57
|
-
this.
|
|
109
|
+
this.clientConfig = withGrokDefaults(config)
|
|
110
|
+
this.client = new OpenAI(this.clientConfig)
|
|
58
111
|
}
|
|
59
112
|
|
|
60
113
|
async generateImages(
|
|
61
114
|
options: ImageGenerationOptions<GrokImageProviderOptions>,
|
|
62
115
|
): Promise<ImageGenerationResult> {
|
|
63
|
-
const { model,
|
|
116
|
+
const { model, numberOfImages, size, modelOptions } = options
|
|
117
|
+
|
|
118
|
+
const resolved = resolveMediaPrompt(options.prompt)
|
|
119
|
+
const prompt = resolved.text
|
|
120
|
+
|
|
121
|
+
if (resolved.videos.length > 0 || resolved.audios.length > 0) {
|
|
122
|
+
throw new Error(
|
|
123
|
+
`grok.generateImages does not support video / audio prompt parts on model ${model}.`,
|
|
124
|
+
)
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
if (resolved.images.length > 0) {
|
|
128
|
+
if (!isGrokImagineImageModel(model)) {
|
|
129
|
+
throw new Error(
|
|
130
|
+
`grok: model "${model}" does not support image prompt parts. ` +
|
|
131
|
+
`Image-conditioned generation requires an Imagine API model ` +
|
|
132
|
+
`('grok-imagine-image' or 'grok-imagine-image-quality').`,
|
|
133
|
+
)
|
|
134
|
+
}
|
|
135
|
+
return await this.editImages(options, resolved)
|
|
136
|
+
}
|
|
64
137
|
|
|
65
138
|
validatePrompt({ prompt, model })
|
|
66
139
|
validateImageSize(model, size)
|
|
67
140
|
validateNumberOfImages(model, numberOfImages)
|
|
68
141
|
|
|
69
|
-
|
|
70
|
-
|
|
142
|
+
// grok-imagine models are aspect-ratio sized: the generic `size` option
|
|
143
|
+
// carries an "aspectRatio_resolution" template (e.g. '16:9_2k', like
|
|
144
|
+
// Gemini native image models) and maps to the Imagine API's
|
|
145
|
+
// `aspect_ratio` / `resolution` parameters instead of OpenAI-style `size`.
|
|
146
|
+
const isImagine = isGrokImagineImageModel(model)
|
|
147
|
+
const request = {
|
|
71
148
|
model,
|
|
72
149
|
prompt,
|
|
73
150
|
n: numberOfImages ?? 1,
|
|
74
|
-
...(
|
|
151
|
+
...(isImagine
|
|
152
|
+
? imagineSizeParams(size)
|
|
153
|
+
: size !== undefined && {
|
|
154
|
+
size: size,
|
|
155
|
+
}),
|
|
75
156
|
stream: false,
|
|
76
157
|
...modelOptions,
|
|
77
|
-
}
|
|
158
|
+
} as OpenAI_SDK.Images.ImageGenerateParamsNonStreaming
|
|
78
159
|
|
|
79
160
|
try {
|
|
80
161
|
options.logger.request(
|
|
@@ -122,6 +203,106 @@ export class GrokImageAdapter<
|
|
|
122
203
|
throw error
|
|
123
204
|
}
|
|
124
205
|
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* Image-conditioned generation via xAI's Imagine API.
|
|
209
|
+
*
|
|
210
|
+
* The `/v1/images/edits` endpoint takes `application/json` (the OpenAI
|
|
211
|
+
* SDK's `images.edit()` sends `multipart/form-data`, which xAI rejects),
|
|
212
|
+
* so this path issues the request directly. One input is sent as
|
|
213
|
+
* `image: { url }`; multiple inputs (up to 3) as `images: [{ url }, ...]`,
|
|
214
|
+
* addressed by xAI in the order they are sent. The prompt text is sent
|
|
215
|
+
* verbatim — no referencing markers are injected.
|
|
216
|
+
*/
|
|
217
|
+
private async editImages(
|
|
218
|
+
options: ImageGenerationOptions<GrokImageProviderOptions>,
|
|
219
|
+
resolved: ResolvedMediaPrompt,
|
|
220
|
+
): Promise<ImageGenerationResult> {
|
|
221
|
+
const { model, numberOfImages, size, modelOptions, logger } = options
|
|
222
|
+
const prompt = resolved.text
|
|
223
|
+
const imageInputs = resolved.images
|
|
224
|
+
|
|
225
|
+
const unsupportedRole = imageInputs.find(
|
|
226
|
+
(part) =>
|
|
227
|
+
part.metadata?.role === 'mask' || part.metadata?.role === 'control',
|
|
228
|
+
)
|
|
229
|
+
if (unsupportedRole) {
|
|
230
|
+
throw new Error(
|
|
231
|
+
`grok: the Imagine API has no ${unsupportedRole.metadata?.role} input; ` +
|
|
232
|
+
`only source/reference images are supported.`,
|
|
233
|
+
)
|
|
234
|
+
}
|
|
235
|
+
if (imageInputs.length > MAX_EDIT_IMAGES) {
|
|
236
|
+
throw new Error(
|
|
237
|
+
`grok: model "${model}" accepts at most ${MAX_EDIT_IMAGES} source images; received ${imageInputs.length}.`,
|
|
238
|
+
)
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
validatePrompt({ prompt, model })
|
|
242
|
+
validateImageSize(model, size)
|
|
243
|
+
validateNumberOfImages(model, numberOfImages)
|
|
244
|
+
|
|
245
|
+
const urls = imageInputs.map((part) => imagePartToUrl(part))
|
|
246
|
+
const request: Record<string, unknown> = {
|
|
247
|
+
model,
|
|
248
|
+
prompt,
|
|
249
|
+
...(urls.length === 1
|
|
250
|
+
? { image: { url: urls[0] } }
|
|
251
|
+
: { images: urls.map((url) => ({ url })) }),
|
|
252
|
+
...(numberOfImages !== undefined && { n: numberOfImages }),
|
|
253
|
+
...imagineSizeParams(size),
|
|
254
|
+
...modelOptions,
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
try {
|
|
258
|
+
logger.request(
|
|
259
|
+
`activity=image provider=${this.name} model=${model} edit images=${urls.length}`,
|
|
260
|
+
{ provider: this.name, model },
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
const response = await fetch(
|
|
264
|
+
`${this.clientConfig.baseURL}/images/edits`,
|
|
265
|
+
{
|
|
266
|
+
method: 'POST',
|
|
267
|
+
headers: {
|
|
268
|
+
'Content-Type': 'application/json',
|
|
269
|
+
Authorization: `Bearer ${this.clientConfig.apiKey}`,
|
|
270
|
+
},
|
|
271
|
+
body: JSON.stringify(request),
|
|
272
|
+
},
|
|
273
|
+
)
|
|
274
|
+
if (!response.ok) {
|
|
275
|
+
const body = await response.text()
|
|
276
|
+
throw new Error(
|
|
277
|
+
`grok: image edit request failed (${response.status} ${response.statusText}): ${body}`,
|
|
278
|
+
)
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
const result = (await response.json()) as GrokImageEditResponse
|
|
282
|
+
const images: Array<GeneratedImage> = (result.data ?? []).flatMap(
|
|
283
|
+
(item): Array<GeneratedImage> => {
|
|
284
|
+
if (item.b64_json) return [{ b64Json: item.b64_json }]
|
|
285
|
+
if (item.url) return [{ url: item.url }]
|
|
286
|
+
return []
|
|
287
|
+
},
|
|
288
|
+
)
|
|
289
|
+
if (images.length === 0) {
|
|
290
|
+
throw new Error('grok: image edit response contained no images')
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
return {
|
|
294
|
+
id: generateId(this.name),
|
|
295
|
+
model,
|
|
296
|
+
images,
|
|
297
|
+
}
|
|
298
|
+
} catch (error: unknown) {
|
|
299
|
+
logger.errors(`${this.name}.generateImages fatal`, {
|
|
300
|
+
error: toRunErrorPayload(error, `${this.name}.generateImages failed`),
|
|
301
|
+
source: `${this.name}.generateImages`,
|
|
302
|
+
})
|
|
303
|
+
throw error
|
|
304
|
+
}
|
|
305
|
+
}
|
|
125
306
|
}
|
|
126
307
|
|
|
127
308
|
/**
|
|
@@ -10,6 +10,84 @@
|
|
|
10
10
|
*/
|
|
11
11
|
export type GrokImageSize = '1024x1024' | '1536x1024' | '1024x1536'
|
|
12
12
|
|
|
13
|
+
/**
|
|
14
|
+
* Aspect ratios accepted by the grok-imagine image models.
|
|
15
|
+
*/
|
|
16
|
+
export type GrokImagineAspectRatio =
|
|
17
|
+
| '1:1'
|
|
18
|
+
| '3:4'
|
|
19
|
+
| '4:3'
|
|
20
|
+
| '9:16'
|
|
21
|
+
| '16:9'
|
|
22
|
+
| '2:3'
|
|
23
|
+
| '3:2'
|
|
24
|
+
| '9:19.5'
|
|
25
|
+
| '19.5:9'
|
|
26
|
+
| '9:20'
|
|
27
|
+
| '20:9'
|
|
28
|
+
| '1:2'
|
|
29
|
+
| '2:1'
|
|
30
|
+
| 'auto'
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Resolution tiers for the grok-imagine image models.
|
|
34
|
+
*/
|
|
35
|
+
export type GrokImagineResolution = '1k' | '2k'
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Size strings for grok-imagine image models. The Imagine API is
|
|
39
|
+
* aspect-ratio based rather than pixel-size based; like Gemini's native
|
|
40
|
+
* image models, the generic `size` option uses an
|
|
41
|
+
* `aspectRatio_resolution` template ("16:9_2k") — the resolution suffix is
|
|
42
|
+
* optional ("16:9" uses the API default of 1k).
|
|
43
|
+
*/
|
|
44
|
+
export type GrokImagineImageSize =
|
|
45
|
+
| GrokImagineAspectRatio
|
|
46
|
+
| `${GrokImagineAspectRatio}_${GrokImagineResolution}`
|
|
47
|
+
|
|
48
|
+
const GROK_IMAGINE_ASPECT_RATIOS: ReadonlyArray<string> = [
|
|
49
|
+
'1:1',
|
|
50
|
+
'3:4',
|
|
51
|
+
'4:3',
|
|
52
|
+
'9:16',
|
|
53
|
+
'16:9',
|
|
54
|
+
'2:3',
|
|
55
|
+
'3:2',
|
|
56
|
+
'9:19.5',
|
|
57
|
+
'19.5:9',
|
|
58
|
+
'9:20',
|
|
59
|
+
'20:9',
|
|
60
|
+
'1:2',
|
|
61
|
+
'2:1',
|
|
62
|
+
'auto',
|
|
63
|
+
]
|
|
64
|
+
|
|
65
|
+
const GROK_IMAGINE_RESOLUTIONS: ReadonlyArray<string> = ['1k', '2k']
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Models served by xAI's Imagine API. They are aspect-ratio sized and
|
|
69
|
+
* support image-conditioned generation via `/v1/images/edits`; the legacy
|
|
70
|
+
* grok-2-image-1212 model is pixel-sized and text-to-image only.
|
|
71
|
+
*/
|
|
72
|
+
export function isGrokImagineImageModel(model: string): boolean {
|
|
73
|
+
return model.startsWith('grok-imagine-image')
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Parses a grok-imagine size string into its components.
|
|
78
|
+
* Format: "aspectRatio" or "aspectRatio_resolution",
|
|
79
|
+
* e.g. "16:9_2k" → { aspectRatio: "16:9", resolution: "2k" }.
|
|
80
|
+
* Returns undefined when the string doesn't match the template.
|
|
81
|
+
*/
|
|
82
|
+
export function parseGrokImagineSize(
|
|
83
|
+
size: string,
|
|
84
|
+
): { aspectRatio: string; resolution?: string } | undefined {
|
|
85
|
+
const match = size.match(/^([\d.]+:[\d.]+|auto)(?:_(.+))?$/)
|
|
86
|
+
const [, aspectRatio, resolution] = match ?? []
|
|
87
|
+
if (aspectRatio === undefined) return undefined
|
|
88
|
+
return { aspectRatio, ...(resolution !== undefined && { resolution }) }
|
|
89
|
+
}
|
|
90
|
+
|
|
13
91
|
/**
|
|
14
92
|
* Base provider options for Grok image models
|
|
15
93
|
*/
|
|
@@ -39,11 +117,37 @@ export interface GrokImageProviderOptions extends GrokImageBaseProviderOptions {
|
|
|
39
117
|
response_format?: 'url' | 'b64_json'
|
|
40
118
|
}
|
|
41
119
|
|
|
120
|
+
/**
|
|
121
|
+
* Provider options for the grok-imagine image models (generation and
|
|
122
|
+
* image-conditioned editing via xAI's Imagine API).
|
|
123
|
+
*/
|
|
124
|
+
export interface GrokImagineImageProviderOptions extends GrokImageBaseProviderOptions {
|
|
125
|
+
/**
|
|
126
|
+
* The format in which generated images are returned.
|
|
127
|
+
* @default 'url'
|
|
128
|
+
*/
|
|
129
|
+
response_format?: 'url' | 'b64_json'
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Output resolution.
|
|
133
|
+
* @default '1k'
|
|
134
|
+
*/
|
|
135
|
+
resolution?: '1k' | '2k'
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Processing tier for the request.
|
|
139
|
+
* @default 'default'
|
|
140
|
+
*/
|
|
141
|
+
service_tier?: 'default' | 'priority'
|
|
142
|
+
}
|
|
143
|
+
|
|
42
144
|
/**
|
|
43
145
|
* Type-only map from model name to its specific provider options.
|
|
44
146
|
*/
|
|
45
147
|
export type GrokImageModelProviderOptionsByName = {
|
|
46
148
|
'grok-2-image-1212': GrokImageProviderOptions
|
|
149
|
+
'grok-imagine-image': GrokImagineImageProviderOptions
|
|
150
|
+
'grok-imagine-image-quality': GrokImagineImageProviderOptions
|
|
47
151
|
}
|
|
48
152
|
|
|
49
153
|
/**
|
|
@@ -51,6 +155,19 @@ export type GrokImageModelProviderOptionsByName = {
|
|
|
51
155
|
*/
|
|
52
156
|
export type GrokImageModelSizeByName = {
|
|
53
157
|
'grok-2-image-1212': GrokImageSize
|
|
158
|
+
'grok-imagine-image': GrokImagineImageSize
|
|
159
|
+
'grok-imagine-image-quality': GrokImagineImageSize
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Per-model prompt input modalities. Imagine API models accept image parts
|
|
164
|
+
* in the prompt (routed to `/v1/images/edits`, up to 3 images, addressed by
|
|
165
|
+
* xAI in request order); grok-2-image is text-to-image only.
|
|
166
|
+
*/
|
|
167
|
+
export type GrokImageModelInputModalitiesByName = {
|
|
168
|
+
'grok-2-image-1212': readonly []
|
|
169
|
+
'grok-imagine-image': readonly ['image']
|
|
170
|
+
'grok-imagine-image-quality': readonly ['image']
|
|
54
171
|
}
|
|
55
172
|
|
|
56
173
|
/**
|
|
@@ -71,6 +188,23 @@ export function validateImageSize(
|
|
|
71
188
|
): void {
|
|
72
189
|
if (!size) return
|
|
73
190
|
|
|
191
|
+
if (isGrokImagineImageModel(model)) {
|
|
192
|
+
const parsed = parseGrokImagineSize(size)
|
|
193
|
+
if (
|
|
194
|
+
!parsed ||
|
|
195
|
+
!GROK_IMAGINE_ASPECT_RATIOS.includes(parsed.aspectRatio) ||
|
|
196
|
+
(parsed.resolution !== undefined &&
|
|
197
|
+
!GROK_IMAGINE_RESOLUTIONS.includes(parsed.resolution))
|
|
198
|
+
) {
|
|
199
|
+
throw new Error(
|
|
200
|
+
`Size "${size}" is not supported by model "${model}". ` +
|
|
201
|
+
`Expected an aspect ratio (${GROK_IMAGINE_ASPECT_RATIOS.join(', ')}) ` +
|
|
202
|
+
`optionally suffixed with a resolution ("16:9_2k"; resolutions: ${GROK_IMAGINE_RESOLUTIONS.join(', ')}).`,
|
|
203
|
+
)
|
|
204
|
+
}
|
|
205
|
+
return
|
|
206
|
+
}
|
|
207
|
+
|
|
74
208
|
const validSizes: Record<string, Array<string>> = {
|
|
75
209
|
'grok-2-image-1212': ['1024x1024', '1536x1024', '1024x1536'],
|
|
76
210
|
}
|
package/src/model-meta.ts
CHANGED
|
@@ -219,6 +219,39 @@ const GROK_2_IMAGE = {
|
|
|
219
219
|
},
|
|
220
220
|
} as const satisfies ModelMeta
|
|
221
221
|
|
|
222
|
+
// Imagine API image models. Pricing is per generated image (output only).
|
|
223
|
+
const GROK_IMAGINE_IMAGE = {
|
|
224
|
+
name: 'grok-imagine-image',
|
|
225
|
+
supports: {
|
|
226
|
+
input: ['text', 'image'],
|
|
227
|
+
output: ['image'],
|
|
228
|
+
},
|
|
229
|
+
pricing: {
|
|
230
|
+
input: {
|
|
231
|
+
normal: 0,
|
|
232
|
+
},
|
|
233
|
+
output: {
|
|
234
|
+
normal: 0.02,
|
|
235
|
+
},
|
|
236
|
+
},
|
|
237
|
+
} as const satisfies ModelMeta
|
|
238
|
+
|
|
239
|
+
const GROK_IMAGINE_IMAGE_QUALITY = {
|
|
240
|
+
name: 'grok-imagine-image-quality',
|
|
241
|
+
supports: {
|
|
242
|
+
input: ['text', 'image'],
|
|
243
|
+
output: ['image'],
|
|
244
|
+
},
|
|
245
|
+
pricing: {
|
|
246
|
+
input: {
|
|
247
|
+
normal: 0,
|
|
248
|
+
},
|
|
249
|
+
output: {
|
|
250
|
+
normal: 0.05,
|
|
251
|
+
},
|
|
252
|
+
},
|
|
253
|
+
} as const satisfies ModelMeta
|
|
254
|
+
|
|
222
255
|
/**
|
|
223
256
|
* Grok Chat Models
|
|
224
257
|
* Based on xAI's available models as of 2025
|
|
@@ -349,7 +382,11 @@ export const GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS = new Set<string>([
|
|
|
349
382
|
/**
|
|
350
383
|
* Grok Image Generation Models
|
|
351
384
|
*/
|
|
352
|
-
export const GROK_IMAGE_MODELS = [
|
|
385
|
+
export const GROK_IMAGE_MODELS = [
|
|
386
|
+
GROK_2_IMAGE.name,
|
|
387
|
+
GROK_IMAGINE_IMAGE.name,
|
|
388
|
+
GROK_IMAGINE_IMAGE_QUALITY.name,
|
|
389
|
+
] as const
|
|
353
390
|
|
|
354
391
|
// xAI's `/v1/tts` endpoint is endpoint-addressed and does not take a `model`
|
|
355
392
|
// parameter. This synthetic identifier satisfies the SDK's `TTSOptions.model`
|