@tanstack/ai-grok 0.18.5 → 0.18.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -150,7 +150,6 @@ var GrokVideoAdapter = class extends BaseVideoAdapter {
150
150
  if (referenceAudioCount > 3) throw new Error(`${this.name}: ${model} accepts at most 3 reference voices; received ${referenceAudioCount}.`);
151
151
  if (referenceImageCount > 7) throw new Error(`${this.name}: ${model} accepts at most 7 reference images; received ${referenceImageCount}.`);
152
152
  const [startFrame] = startFrames;
153
- if (startFrame && hasReference) throw new Error(`${this.name}: image-to-video and reference-to-video cannot be combined. Use a starting-frame image, or reference images / voices, not both.`);
154
153
  const parsedSize = size !== void 0 ? parseGrokVideoSize(size) : void 0;
155
154
  const resolvedResolution = generationOptions.resolution ?? parsedSize?.resolution;
156
155
  if (hasReference && resolvedResolution === "1080p") throw new Error(`${this.name}: reference-to-video is capped at 720p on ${model}.`);
@@ -1 +1 @@
1
- {"version":3,"file":"video.js","names":[],"sources":["../../../src/adapters/video.ts"],"sourcesContent":["import { resolveMediaPrompt } from '@tanstack/ai'\nimport { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'\nimport {\n GROK_VIDEO_MAX_REFERENCE_AUDIOS,\n GROK_VIDEO_MAX_REFERENCE_IMAGES,\n getGrokVideoDurationOptions,\n isGrokVideoReferenceModel,\n isGrokVideoSourceModel,\n parseGrokVideoSize,\n validateVideoSize,\n} from '../video/video-provider-options'\nimport type { DurationOptions } from '@tanstack/ai/adapters'\nimport type {\n ImagePart,\n MediaInputMetadata,\n TokenUsage,\n VideoGenerationOptions,\n VideoJobResult,\n VideoPart,\n VideoStatusResult,\n VideoUrlResult,\n} from '@tanstack/ai'\nimport type { GrokVideoModel } from '../model-meta'\nimport type {\n GrokVideoModelDurationByName,\n GrokVideoModelInputModalitiesByName,\n GrokVideoModelProviderOptionsByName,\n GrokVideoModelSizeByName,\n GrokVideoRuntimeOptions,\n} from '../video/video-provider-options'\nimport type { GrokClientConfig } from '../utils/client'\n\n/**\n * Configuration for Grok video adapter.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GrokVideoConfig extends GrokClientConfig {}\n\n/**\n * xAI bills video generation in \"USD ticks\": 10^10 ticks per US dollar\n * (e.g. one grok-imagine-video-1.5 second costs $0.08 = 800_000_000 ticks).\n */\nconst USD_TICKS_PER_DOLLAR = 10_000_000_000\n\n/** Response of the POST /v1/videos/{generations,edits,extensions} endpoints. */\ninterface GrokVideoCreateResponse {\n request_id?: string\n}\n\n/** Response of GET /v1/videos/{request_id}. */\ninterface GrokVideoStatusResponse {\n status?: string\n progress?: number\n model?: string\n video?: {\n url?: string\n duration?: number\n }\n usage?: {\n cost_in_usd_ticks?: number\n }\n error?: string\n}\n\n/**\n * Convert a TanStack image / video part to the URL string accepted by xAI's\n * Imagine video endpoints: public URLs pass through (fetched by xAI's\n * servers), data sources become base64 data URIs.\n */\nfunction mediaPartToUrl(\n part: ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata>,\n): string {\n if (part.source.type === 'url') return part.source.value\n return `data:${part.source.mimeType};base64,${part.source.value}`\n}\n\nfunction buildGrokVideoUsage(\n response: GrokVideoStatusResponse,\n): TokenUsage | undefined {\n const seconds = response.video?.duration\n const ticks = response.usage?.cost_in_usd_ticks\n if (seconds === undefined && ticks === undefined) return undefined\n return {\n promptTokens: 0,\n completionTokens: 0,\n totalTokens: 0,\n ...(seconds !== undefined && {\n billed: { quantity: seconds, unit: 'seconds' },\n unitsBilled: seconds,\n }),\n ...(ticks !== undefined && { cost: ticks / USD_TICKS_PER_DOLLAR }),\n }\n}\n\n/**\n * Grok Video Generation Adapter (xAI Imagine API)\n *\n * Tree-shakeable adapter for the grok-imagine video models using the\n * async jobs/polling architecture: create a generation request, poll it,\n * then read the completed video URL.\n *\n * Both models support text-to-video and image-to-video;\n * `grok-imagine-video-1.5` is xAI's documented default and adds native\n * 1080p generation plus reference-to-video inputs. Source-video edit\n * and extend are `grok-imagine-video` only.\n *\n * The Imagine video endpoints are not part of the OpenAI SDK surface (and\n * xAI rejects the SDK's multipart paths), so requests are plain JSON calls\n * issued with the configured `fetch` (or the global one).\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * Features:\n * - Async job-based video generation (1–15 second clips with audio)\n * - Aspect-ratio sizing via the \"aspectRatio_resolution\" size template\n * (e.g. '16:9_720p'), consistent with the grok-imagine image models\n * - Image-to-video via an `image` prompt part (starting frame URL or data URI)\n * - Reference-to-video via image prompt parts with\n * `metadata.role: 'reference'` or `'character'` (→ `reference_images`)\n * and preset voices via `modelOptions.reference_audios`\n * (grok-imagine-video-1.5 only)\n * - Video editing / extension on `grok-imagine-video` via a source\n * `video` prompt part and `modelOptions.mode: 'edit' | 'extend'`\n * (`/v1/videos/edits` / `/v1/videos/extensions`; in extend mode\n * `duration` is the added tail)\n * - Usage reporting: billed seconds (`usage.billed`) and exact cost\n */\nexport class GrokVideoAdapter<\n TModel extends GrokVideoModel,\n> extends BaseVideoAdapter<\n TModel,\n GrokVideoModelProviderOptionsByName[TModel],\n GrokVideoModelProviderOptionsByName,\n GrokVideoModelSizeByName,\n GrokVideoModelInputModalitiesByName,\n GrokVideoModelDurationByName\n> {\n readonly name = 'grok' as const\n\n private readonly clientConfig: GrokVideoConfig\n\n constructor(config: GrokVideoConfig, model: TModel) {\n super({}, model)\n this.clientConfig = withGrokDefaults(config)\n }\n\n private async request(\n path: string,\n init?: Omit<RequestInit, 'headers'>,\n ): Promise<Response> {\n // workerd's fetch must be invoked with this === globalThis\n const fetchFn = this.clientConfig.fetch ?? globalThis.fetch.bind(globalThis)\n return await fetchFn(`${this.clientConfig.baseURL}${path}`, {\n ...init,\n headers: {\n 'Content-Type': 'application/json',\n Authorization: `Bearer ${this.clientConfig.apiKey}`,\n },\n })\n }\n\n /**\n * Reads the error message out of an Imagine API error body\n * (`{\"code\": \"...\", \"error\": \"...\"}`), falling back to the raw text.\n */\n private async errorMessage(response: Response): Promise<string> {\n const body = await response.text()\n try {\n const parsed: unknown = JSON.parse(body)\n if (\n typeof parsed === 'object' &&\n parsed !== null &&\n 'error' in parsed &&\n typeof parsed.error === 'string'\n ) {\n return parsed.error\n }\n } catch {\n // not JSON — fall through to the raw body\n }\n return body\n }\n\n async createVideoJob(\n options: VideoGenerationOptions<\n GrokVideoModelProviderOptionsByName[TModel],\n GrokVideoModelSizeByName[TModel],\n GrokVideoModelDurationByName[TModel]\n >,\n ): Promise<VideoJobResult> {\n const { model, size, modelOptions, logger } = options\n\n // `mode` is a routing hint for this adapter, not an API field — strip it\n // before the remaining options are spread onto the request body. The\n // per-model map narrows what callers can pass, but modelOptions often\n // arrives as deserialized JSON, so the adapter handles the widest option\n // surface (the 1.5 shape) uniformly and gates by model at runtime.\n const { mode, ...wireOptions } = (modelOptions ??\n {}) as GrokVideoRuntimeOptions\n\n // `mode` is typed 'edit' | 'extend' but reaches us untrusted from JSON\n // callers. An unrecognised value must not fall through to the\n // generations endpoint with a source-video body — that would silently\n // run (and bill) a generation the caller never asked for.\n if (mode !== undefined && mode !== 'edit' && mode !== 'extend') {\n throw new Error(\n `${this.name}: unknown modelOptions.mode '${String(mode)}'. ` +\n `Expected 'edit' or 'extend'.`,\n )\n }\n\n // The interleaved prompt decomposes into verbatim text plus typed media\n // buckets. Reference audio is voice-id based (not an audio file), so\n // audio prompt parts have no request field to land in.\n const resolved = resolveMediaPrompt(options.prompt)\n if (resolved.audios.length > 0) {\n throw new Error(\n `${this.name}.createVideoJob does not support audio prompt parts (model: ${model}). ` +\n `To reference a preset voice, pass modelOptions.reference_audios ` +\n `(e.g. [{ voice_id: 'eve' }]).`,\n )\n }\n\n // A video prompt part is the source clip for edit / extension mode.\n // Those endpoints are grok-imagine-video only — 1.5 has no video input.\n if (\n !isGrokVideoSourceModel(model) &&\n (mode !== undefined || resolved.videos.length > 0)\n ) {\n throw new Error(\n `${this.name}: ${model} does not support video editing or extension. ` +\n `Use 'grok-imagine-video' for /v1/videos/edits and /v1/videos/extensions.`,\n )\n }\n\n // The mode must be chosen explicitly because the two endpoints have\n // different semantics (edit rewrites the clip, extend appends\n // `duration` seconds).\n if (resolved.videos.length > 1) {\n throw new Error(\n `${this.name}: ${model} accepts at most one source video; received ${resolved.videos.length}.`,\n )\n }\n const [sourceVideo] = resolved.videos\n if (sourceVideo && mode === undefined) {\n throw new Error(\n `${this.name}: a video prompt part needs modelOptions.mode set to ` +\n `'edit' (rewrite the clip) or 'extend' (append to it).`,\n )\n }\n if (!sourceVideo && mode !== undefined) {\n throw new Error(\n `${this.name}: modelOptions.mode '${mode}' requires a video prompt ` +\n `part carrying the source clip.`,\n )\n }\n\n if (mode !== undefined && sourceVideo) {\n return await this.createSourceVideoJob({\n model,\n mode,\n sourceVideo,\n resolved,\n wireOptions,\n size,\n genericDuration: options.duration,\n logger,\n })\n }\n\n validateVideoSize(model, size)\n\n // Pull the specially-handled keys out of the wire options: `duration`\n // is folded into the snapped value below, and the reference fields are\n // re-added explicitly so a JSON-serialized `null` or empty array reads\n // as \"unset\" instead of leaking onto the wire.\n const {\n duration: rawOptionDuration,\n reference_images: explicitReferenceImages,\n reference_audios: referenceAudios,\n ...generationOptions\n } = wireOptions\n\n // Coerce the requested duration into the model's valid range (1–15s,\n // integer) instead of rejecting it — `snapDuration` clamps and rounds.\n // modelOptions wins over the generic `duration`, mirroring the size\n // precedence below.\n const rawDuration = rawOptionDuration ?? options.duration\n const duration =\n rawDuration != null ? this.snapDuration(rawDuration) : undefined\n\n // Image parts split by role: un-roled / 'start_frame' images become the\n // starting frame (image-to-video); 'reference' / 'character' images\n // become reference_images (reference-to-video). The Imagine API has no\n // mask / control / end-frame inputs. Unknown role strings (possible via\n // JSON callers) throw rather than silently dropping the part.\n const startFrames: Array<ImagePart<MediaInputMetadata>> = []\n const referenceImages: Array<{ url: string }> = []\n for (const part of resolved.images) {\n const role = part.metadata?.role\n switch (role) {\n case 'mask':\n case 'control':\n case 'end_frame':\n throw new Error(\n `${this.name}: the Imagine video API has no '${role}' image ` +\n `input on model ${model}. Use an un-roled / 'start_frame' ` +\n `image as the starting frame, or 'reference' images.`,\n )\n case 'reference':\n case 'character':\n referenceImages.push({ url: mediaPartToUrl(part) })\n break\n case 'start_frame':\n case undefined:\n startFrames.push(part)\n break\n default:\n throw new Error(\n `${this.name}: unknown image metadata.role '${String(role)}'. ` +\n `Expected 'start_frame', 'reference', or 'character'.`,\n )\n }\n }\n if (startFrames.length > 1) {\n throw new Error(\n `${this.name}: ${model} accepts at most one starting-frame image; received ${startFrames.length}. ` +\n `Use metadata.role: 'reference' for reference-to-video inputs.`,\n )\n }\n // Explicit modelOptions.reference_images replaces the part-derived list\n // (an explicit empty array means \"none\").\n const finalReferenceImages =\n explicitReferenceImages ??\n (referenceImages.length > 0 ? referenceImages : undefined)\n const referenceImageCount = finalReferenceImages?.length ?? 0\n const referenceAudioCount = referenceAudios?.length ?? 0\n const hasReference = referenceImageCount > 0 || referenceAudioCount > 0\n\n // Reference inputs are a grok-imagine-video-1.5 feature. The per-model\n // options map already hides the fields from other models at compile\n // time; this runtime gate covers prompt-part roles and untyped callers.\n if (!isGrokVideoReferenceModel(model) && hasReference) {\n throw new Error(\n `${this.name}: ${model} does not support reference-to-video inputs. ` +\n `Use 'grok-imagine-video-1.5' for reference_images / reference_audios.`,\n )\n }\n if (referenceAudioCount > GROK_VIDEO_MAX_REFERENCE_AUDIOS) {\n throw new Error(\n `${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_AUDIOS} reference voices; received ${referenceAudioCount}.`,\n )\n }\n if (referenceImageCount > GROK_VIDEO_MAX_REFERENCE_IMAGES) {\n throw new Error(\n `${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_IMAGES} reference images; received ${referenceImageCount}.`,\n )\n }\n\n // Image-to-video: the single image prompt part becomes the starting frame\n // and the prompt text describes the desired motion. URL sources are\n // fetched by xAI's servers; data sources are sent as base64 data URIs.\n const [startFrame] = startFrames\n\n // xAI rejects `image` + `reference_images` / `reference_audios` as a\n // 400: only one of image-to-video or reference-to-video can be active.\n if (startFrame && hasReference) {\n throw new Error(\n `${this.name}: image-to-video and reference-to-video cannot be combined. ` +\n `Use a starting-frame image, or reference images / voices, not both.`,\n )\n }\n\n // The generic `size` option carries an \"aspectRatio_resolution\" template\n // (e.g. '16:9_720p') and maps to the Imagine API's `aspect_ratio` /\n // `resolution` parameters; explicit modelOptions win over the template\n // (including `reference_images`, which replaces the part-derived list).\n const parsedSize = size !== undefined ? parseGrokVideoSize(size) : undefined\n const resolvedResolution =\n generationOptions.resolution ?? parsedSize?.resolution\n if (hasReference && resolvedResolution === '1080p') {\n throw new Error(\n `${this.name}: reference-to-video is capped at 720p on ${model}.`,\n )\n }\n const request = {\n model,\n prompt: resolved.text,\n ...(startFrame && { image: { url: mediaPartToUrl(startFrame) } }),\n ...(referenceImageCount > 0 && {\n reference_images: finalReferenceImages,\n }),\n ...(referenceAudioCount > 0 && {\n reference_audios: referenceAudios,\n }),\n ...(parsedSize && {\n aspect_ratio: parsedSize.aspectRatio,\n ...(parsedSize.resolution !== undefined && {\n resolution: parsedSize.resolution,\n }),\n }),\n // The remaining options spread after the size template so explicit\n // aspect_ratio / resolution win over it; duration and the reference\n // fields were destructured out above and re-added normalized.\n ...generationOptions,\n ...(duration !== undefined && { duration }),\n }\n\n return await this.postVideoJob('/videos/generations', request, {\n model,\n logger,\n logLine: `activity=video.create provider=${this.name} model=${model} mode=generate size=${size ?? 'default'} duration=${duration ?? 'default'}`,\n })\n }\n\n /**\n * Build and post an edit / extension request. Both endpoints take only\n * `model`, `prompt`, and the source `video` (plus `duration` — the length\n * of the added tail — for extensions): output geometry is inherited from\n * the source clip, capped at 720p, and edit outputs also inherit the\n * source length. Rather than sending fields the API documents as ignored,\n * the inapplicable options are rejected with actionable errors.\n */\n private async createSourceVideoJob(args: {\n model: string\n mode: 'edit' | 'extend'\n sourceVideo: VideoPart<MediaInputMetadata>\n resolved: ReturnType<typeof resolveMediaPrompt>\n wireOptions: Omit<GrokVideoRuntimeOptions, 'mode'>\n size: string | undefined\n genericDuration: number | undefined\n logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']\n }): Promise<VideoJobResult> {\n const { model, mode, sourceVideo, resolved, wireOptions, logger } = args\n const endpoint = mode === 'edit' ? '/videos/edits' : '/videos/extensions'\n\n if (resolved.images.length > 0) {\n throw new Error(\n `${this.name}: '${mode}' mode takes only the source video — image ` +\n `prompt parts are not supported by ${endpoint}.`,\n )\n }\n\n // Pull every generation-only key out of the wire options so nothing can\n // leak into the edit/extend body via the spread below. JSON-serialized\n // `null` values (a common \"unset\" encoding) are treated as absent;\n // actual values are rejected with actionable errors.\n const {\n aspect_ratio: aspectRatio,\n resolution,\n duration: modeDuration,\n reference_images: referenceImagesOption,\n reference_audios: referenceAudiosOption,\n ...passthrough\n } = wireOptions\n if (\n (referenceImagesOption?.length ?? 0) > 0 ||\n (referenceAudiosOption?.length ?? 0) > 0\n ) {\n throw new Error(\n `${this.name}: reference inputs are only supported by video ` +\n `generation, not '${mode}' mode.`,\n )\n }\n if (args.size !== undefined || aspectRatio != null || resolution != null) {\n throw new Error(\n `${this.name}: '${mode}' mode does not accept size / aspect_ratio / ` +\n `resolution — the output inherits the source clip's geometry ` +\n `(capped at 720p).`,\n )\n }\n const rawDuration = modeDuration ?? args.genericDuration\n if (mode === 'edit' && rawDuration != null) {\n throw new Error(\n `${this.name}: 'edit' mode does not accept a duration — the output ` +\n `inherits the source clip's length. Use mode 'extend' to append ` +\n `seconds to the clip.`,\n )\n }\n // Extend: the snapped duration is the added-tail length (1–15s).\n const duration =\n rawDuration != null ? this.snapDuration(rawDuration) : undefined\n\n const request = {\n model,\n prompt: resolved.text,\n video: { url: mediaPartToUrl(sourceVideo) },\n ...passthrough,\n ...(duration !== undefined && { duration }),\n }\n\n return await this.postVideoJob(endpoint, request, {\n model,\n logger,\n logLine: `activity=video.create provider=${this.name} model=${model} mode=${mode} duration=${duration ?? 'default'}`,\n })\n }\n\n /**\n * POST a create-job request body to one of the Imagine video endpoints\n * (`/videos/generations`, `/videos/edits`, `/videos/extensions`) and read\n * the `request_id` out of the shared response shape.\n */\n private async postVideoJob(\n endpoint: string,\n request: Record<string, unknown>,\n context: {\n model: string\n logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']\n logLine: string\n },\n ): Promise<VideoJobResult> {\n const { model, logger, logLine } = context\n try {\n logger.request(logLine, { provider: this.name, model })\n\n const response = await this.request(endpoint, {\n method: 'POST',\n body: JSON.stringify(request),\n })\n if (!response.ok) {\n throw new Error(\n `grok: ${endpoint} request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,\n )\n }\n\n const result = (await response.json()) as GrokVideoCreateResponse\n if (!result.request_id) {\n throw new Error(`grok: ${endpoint} response contained no request_id`)\n }\n return { jobId: result.request_id, model }\n } catch (error: unknown) {\n logger.errors(`${this.name}.createVideoJob fatal`, {\n error: toRunErrorPayload(error, `${this.name}.createVideoJob failed`),\n source: `${this.name}.createVideoJob`,\n })\n throw error\n }\n }\n\n private async retrieveJob(jobId: string): Promise<GrokVideoStatusResponse> {\n const response = await this.request(`/videos/${jobId}`)\n if (!response.ok) {\n const error = new Error(\n `grok: video status request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,\n )\n ;(error as { status?: number }).status = response.status\n throw error\n }\n return (await response.json()) as GrokVideoStatusResponse\n }\n\n async getVideoStatus(jobId: string): Promise<VideoStatusResult> {\n let response: GrokVideoStatusResponse\n try {\n response = await this.retrieveJob(jobId)\n } catch (error) {\n if ((error as { status?: number }).status === 404) {\n return { jobId, status: 'failed', error: 'Job not found' }\n }\n throw error\n }\n\n return {\n jobId,\n status: this.mapStatus(response.status),\n ...(response.progress !== undefined && { progress: response.progress }),\n ...(response.error !== undefined && { error: response.error }),\n }\n }\n\n async getVideoUrl(jobId: string): Promise<VideoUrlResult> {\n let response: GrokVideoStatusResponse\n try {\n response = await this.retrieveJob(jobId)\n } catch (error) {\n if ((error as { status?: number }).status === 404) {\n throw new Error(`Video job not found: ${jobId}`)\n }\n throw error\n }\n\n const status = this.mapStatus(response.status)\n if (status === 'failed') {\n throw new Error(\n `Video generation failed${response.error ? `: ${response.error}` : ''}. Job ID: ${jobId}`,\n )\n }\n const url = response.video?.url\n if (!url) {\n throw new Error(\n `Video is not ready for download. Check status first. Job ID: ${jobId}`,\n )\n }\n\n const usage = buildGrokVideoUsage(response)\n return {\n jobId,\n url,\n ...(usage && { usage }),\n }\n }\n\n /**\n * Maps Imagine API job statuses onto the generic video status set. The\n * API reports 'pending' while queued/generating (with a numeric\n * `progress`), then a terminal 'done' / 'failed' / 'expired'.\n */\n protected mapStatus(\n apiStatus: string | undefined,\n ): 'pending' | 'processing' | 'completed' | 'failed' {\n switch (apiStatus) {\n case 'pending':\n case 'queued':\n return 'pending'\n case 'done':\n case 'completed':\n case 'succeeded':\n return 'completed'\n case 'failed':\n case 'expired':\n case 'error':\n case 'cancelled':\n return 'failed'\n case undefined:\n default:\n return 'processing'\n }\n }\n\n /**\n * Both grok-imagine video models accept a continuous 1–15 integer-second\n * range. Consumers can use this to render UI without provider knowledge.\n */\n override availableDurations(): DurationOptions<\n GrokVideoModelDurationByName[TModel]\n > {\n return getGrokVideoDurationOptions(this.model)\n }\n\n /**\n * Coerce a raw seconds value to the closest valid duration (clamped to\n * [1, 15] and rounded to whole seconds).\n */\n override snapDuration(\n seconds: number,\n ): GrokVideoModelDurationByName[TModel] | undefined {\n return snapToDurationOption(seconds, this.availableDurations())\n }\n}\n\n/**\n * Creates a Grok video adapter with an explicit API key.\n * Type resolution happens here at the call site.\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'grok-imagine-video-1.5')\n * @param apiKey - Your xAI API key\n * @param config - Optional additional configuration\n * @returns Configured Grok video adapter instance with resolved types\n *\n * @example\n * ```typescript\n * const adapter = createGrokVideo('grok-imagine-video-1.5', 'xai-...');\n *\n * const { jobId } = await generateVideo({\n * adapter,\n * prompt: 'A beautiful sunset over the ocean',\n * size: '16:9_720p',\n * duration: 5\n * });\n * ```\n */\nexport function createGrokVideo<TModel extends GrokVideoModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GrokVideoConfig, 'apiKey'>,\n): GrokVideoAdapter<TModel> {\n return new GrokVideoAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Grok video adapter with automatic API key detection from environment variables.\n * Type resolution happens here at the call site.\n *\n * Looks for `XAI_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (Browser with injected env)\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'grok-imagine-video-1.5')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Grok video adapter instance with resolved types\n * @throws Error if XAI_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * // Automatically uses XAI_API_KEY from environment\n * const adapter = grokVideo('grok-imagine-video-1.5');\n *\n * // Image-to-video: an optional image prompt part is the starting frame.\n * const { jobId } = await generateVideo({\n * adapter,\n * prompt: [\n * { type: 'text', content: 'Make the cat start playing the piano' },\n * { type: 'image', source: { type: 'url', value: 'https://example.com/cat.png' } },\n * ],\n * });\n *\n * // Poll for status\n * const status = await getVideoJobStatus({ adapter, jobId });\n * ```\n */\nexport function grokVideo<TModel extends GrokVideoModel>(\n model: TModel,\n config?: Omit<GrokVideoConfig, 'apiKey'>,\n): GrokVideoAdapter<TModel> {\n const apiKey = getGrokApiKeyFromEnv()\n return createGrokVideo(model, apiKey, config)\n}\n"],"mappings":";;;;;;;;;;AA6CA,IAAM,uBAAuB;;;;;;AA2B7B,SAAS,eACP,MACQ;CACR,IAAI,KAAK,OAAO,SAAS,OAAO,OAAO,KAAK,OAAO;CACnD,OAAO,QAAQ,KAAK,OAAO,SAAS,UAAU,KAAK,OAAO;AAC5D;AAEA,SAAS,oBACP,UACwB;CACxB,MAAM,UAAU,SAAS,OAAO;CAChC,MAAM,QAAQ,SAAS,OAAO;CAC9B,IAAI,YAAY,KAAA,KAAa,UAAU,KAAA,GAAW,OAAO,KAAA;CACzD,OAAO;EACL,cAAc;EACd,kBAAkB;EAClB,aAAa;EACb,GAAI,YAAY,KAAA,KAAa;GAC3B,QAAQ;IAAE,UAAU;IAAS,MAAM;GAAU;GAC7C,aAAa;EACf;EACA,GAAI,UAAU,KAAA,KAAa,EAAE,MAAM,QAAQ,qBAAqB;CAClE;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmCA,IAAa,mBAAb,cAEU,iBAOR;CACA,OAAgB;CAEhB;CAEA,YAAY,QAAyB,OAAe;EAClD,MAAM,CAAC,GAAG,KAAK;EACf,KAAK,eAAe,iBAAiB,MAAM;CAC7C;CAEA,MAAc,QACZ,MACA,MACmB;EAGnB,OAAO,OADS,KAAK,aAAa,SAAS,WAAW,MAAM,KAAK,UAAU,EAAA,CACtD,GAAG,KAAK,aAAa,UAAU,QAAQ;GAC1D,GAAG;GACH,SAAS;IACP,gBAAgB;IAChB,eAAe,UAAU,KAAK,aAAa;GAC7C;EACF,CAAC;CACH;;;;;CAMA,MAAc,aAAa,UAAqC;EAC9D,MAAM,OAAO,MAAM,SAAS,KAAK;EACjC,IAAI;GACF,MAAM,SAAkB,KAAK,MAAM,IAAI;GACvC,IACE,OAAO,WAAW,YAClB,WAAW,QACX,WAAW,UACX,OAAO,OAAO,UAAU,UAExB,OAAO,OAAO;EAElB,QAAQ,CAER;EACA,OAAO;CACT;CAEA,MAAM,eACJ,SAKyB;EACzB,MAAM,EAAE,OAAO,MAAM,cAAc,WAAW;EAO9C,MAAM,EAAE,MAAM,GAAG,gBAAiB,gBAChC,CAAC;EAMH,IAAI,SAAS,KAAA,KAAa,SAAS,UAAU,SAAS,UACpD,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,+BAA+B,OAAO,IAAI,EAAE,gCAE3D;EAMF,MAAM,WAAW,mBAAmB,QAAQ,MAAM;EAClD,IAAI,SAAS,OAAO,SAAS,GAC3B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,8DAA8D,MAAM,iGAGnF;EAKF,IACE,CAAC,uBAAuB,KAAK,MAC5B,SAAS,KAAA,KAAa,SAAS,OAAO,SAAS,IAEhD,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,uHAEzB;EAMF,IAAI,SAAS,OAAO,SAAS,GAC3B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,8CAA8C,SAAS,OAAO,OAAO,EAC9F;EAEF,MAAM,CAAC,eAAe,SAAS;EAC/B,IAAI,eAAe,SAAS,KAAA,GAC1B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,2GAEf;EAEF,IAAI,CAAC,eAAe,SAAS,KAAA,GAC3B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,uBAAuB,KAAK,yDAE3C;EAGF,IAAI,SAAS,KAAA,KAAa,aACxB,OAAO,MAAM,KAAK,qBAAqB;GACrC;GACA;GACA;GACA;GACA;GACA;GACA,iBAAiB,QAAQ;GACzB;EACF,CAAC;EAGH,kBAAkB,OAAO,IAAI;EAM7B,MAAM,EACJ,UAAU,mBACV,kBAAkB,yBAClB,kBAAkB,iBAClB,GAAG,sBACD;EAMJ,MAAM,cAAc,qBAAqB,QAAQ;EACjD,MAAM,WACJ,eAAe,OAAO,KAAK,aAAa,WAAW,IAAI,KAAA;EAOzD,MAAM,cAAoD,CAAC;EAC3D,MAAM,kBAA0C,CAAC;EACjD,KAAK,MAAM,QAAQ,SAAS,QAAQ;GAClC,MAAM,OAAO,KAAK,UAAU;GAC5B,QAAQ,MAAR;IACE,KAAK;IACL,KAAK;IACL,KAAK,aACH,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,kCAAkC,KAAK,yBAChC,MAAM,sFAE5B;IACF,KAAK;IACL,KAAK;KACH,gBAAgB,KAAK,EAAE,KAAK,eAAe,IAAI,EAAE,CAAC;KAClD;IACF,KAAK;IACL,KAAK,KAAA;KACH,YAAY,KAAK,IAAI;KACrB;IACF,SACE,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,iCAAiC,OAAO,IAAI,EAAE,wDAE7D;GACJ;EACF;EACA,IAAI,YAAY,SAAS,GACvB,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,sDAAsD,YAAY,OAAO,gEAElG;EAIF,MAAM,uBACJ,4BACC,gBAAgB,SAAS,IAAI,kBAAkB,KAAA;EAClD,MAAM,sBAAsB,sBAAsB,UAAU;EAC5D,MAAM,sBAAsB,iBAAiB,UAAU;EACvD,MAAM,eAAe,sBAAsB,KAAK,sBAAsB;EAKtE,IAAI,CAAC,0BAA0B,KAAK,KAAK,cACvC,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,mHAEzB;EAEF,IAAI,sBAAA,GACF,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,gDAAiF,oBAAoB,EAC9H;EAEF,IAAI,sBAAA,GACF,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,gDAAiF,oBAAoB,EAC9H;EAMF,MAAM,CAAC,cAAc;EAIrB,IAAI,cAAc,cAChB,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,gIAEf;EAOF,MAAM,aAAa,SAAS,KAAA,IAAY,mBAAmB,IAAI,IAAI,KAAA;EACnE,MAAM,qBACJ,kBAAkB,cAAc,YAAY;EAC9C,IAAI,gBAAgB,uBAAuB,SACzC,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,4CAA4C,MAAM,EACjE;EAEF,MAAM,UAAU;GACd;GACA,QAAQ,SAAS;GACjB,GAAI,cAAc,EAAE,OAAO,EAAE,KAAK,eAAe,UAAU,EAAE,EAAE;GAC/D,GAAI,sBAAsB,KAAK,EAC7B,kBAAkB,qBACpB;GACA,GAAI,sBAAsB,KAAK,EAC7B,kBAAkB,gBACpB;GACA,GAAI,cAAc;IAChB,cAAc,WAAW;IACzB,GAAI,WAAW,eAAe,KAAA,KAAa,EACzC,YAAY,WAAW,WACzB;GACF;GAIA,GAAG;GACH,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;EAC3C;EAEA,OAAO,MAAM,KAAK,aAAa,uBAAuB,SAAS;GAC7D;GACA;GACA,SAAS,kCAAkC,KAAK,KAAK,SAAS,MAAM,sBAAsB,QAAQ,UAAU,YAAY,YAAY;EACtI,CAAC;CACH;;;;;;;;;CAUA,MAAc,qBAAqB,MASP;EAC1B,MAAM,EAAE,OAAO,MAAM,aAAa,UAAU,aAAa,WAAW;EACpE,MAAM,WAAW,SAAS,SAAS,kBAAkB;EAErD,IAAI,SAAS,OAAO,SAAS,GAC3B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,KAAK,KAAK,+EACgB,SAAS,EAClD;EAOF,MAAM,EACJ,cAAc,aACd,YACA,UAAU,cACV,kBAAkB,uBAClB,kBAAkB,uBAClB,GAAG,gBACD;EACJ,KACG,uBAAuB,UAAU,KAAK,MACtC,uBAAuB,UAAU,KAAK,GAEvC,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,kEACS,KAAK,QAC7B;EAEF,IAAI,KAAK,SAAS,KAAA,KAAa,eAAe,QAAQ,cAAc,MAClE,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,KAAK,KAAK,2HAGzB;EAEF,MAAM,cAAc,gBAAgB,KAAK;EACzC,IAAI,SAAS,UAAU,eAAe,MACpC,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,0IAGf;EAGF,MAAM,WACJ,eAAe,OAAO,KAAK,aAAa,WAAW,IAAI,KAAA;EAEzD,MAAM,UAAU;GACd;GACA,QAAQ,SAAS;GACjB,OAAO,EAAE,KAAK,eAAe,WAAW,EAAE;GAC1C,GAAG;GACH,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;EAC3C;EAEA,OAAO,MAAM,KAAK,aAAa,UAAU,SAAS;GAChD;GACA;GACA,SAAS,kCAAkC,KAAK,KAAK,SAAS,MAAM,QAAQ,KAAK,YAAY,YAAY;EAC3G,CAAC;CACH;;;;;;CAOA,MAAc,aACZ,UACA,SACA,SAKyB;EACzB,MAAM,EAAE,OAAO,QAAQ,YAAY;EACnC,IAAI;GACF,OAAO,QAAQ,SAAS;IAAE,UAAU,KAAK;IAAM;GAAM,CAAC;GAEtD,MAAM,WAAW,MAAM,KAAK,QAAQ,UAAU;IAC5C,QAAQ;IACR,MAAM,KAAK,UAAU,OAAO;GAC9B,CAAC;GACD,IAAI,CAAC,SAAS,IACZ,MAAM,IAAI,MACR,SAAS,SAAS,mBAAmB,SAAS,OAAO,GAAG,SAAS,WAAW,KAAK,MAAM,KAAK,aAAa,QAAQ,GACnH;GAGF,MAAM,SAAU,MAAM,SAAS,KAAK;GACpC,IAAI,CAAC,OAAO,YACV,MAAM,IAAI,MAAM,SAAS,SAAS,kCAAkC;GAEtE,OAAO;IAAE,OAAO,OAAO;IAAY;GAAM;EAC3C,SAAS,OAAgB;GACvB,OAAO,OAAO,GAAG,KAAK,KAAK,wBAAwB;IACjD,OAAO,kBAAkB,OAAO,GAAG,KAAK,KAAK,uBAAuB;IACpE,QAAQ,GAAG,KAAK,KAAK;GACvB,CAAC;GACD,MAAM;EACR;CACF;CAEA,MAAc,YAAY,OAAiD;EACzE,MAAM,WAAW,MAAM,KAAK,QAAQ,WAAW,OAAO;EACtD,IAAI,CAAC,SAAS,IAAI;GAChB,MAAM,wBAAQ,IAAI,MAChB,sCAAsC,SAAS,OAAO,GAAG,SAAS,WAAW,KAAK,MAAM,KAAK,aAAa,QAAQ,GACpH;GACC,MAA+B,SAAS,SAAS;GAClD,MAAM;EACR;EACA,OAAQ,MAAM,SAAS,KAAK;CAC9B;CAEA,MAAM,eAAe,OAA2C;EAC9D,IAAI;EACJ,IAAI;GACF,WAAW,MAAM,KAAK,YAAY,KAAK;EACzC,SAAS,OAAO;GACd,IAAK,MAA8B,WAAW,KAC5C,OAAO;IAAE;IAAO,QAAQ;IAAU,OAAO;GAAgB;GAE3D,MAAM;EACR;EAEA,OAAO;GACL;GACA,QAAQ,KAAK,UAAU,SAAS,MAAM;GACtC,GAAI,SAAS,aAAa,KAAA,KAAa,EAAE,UAAU,SAAS,SAAS;GACrE,GAAI,SAAS,UAAU,KAAA,KAAa,EAAE,OAAO,SAAS,MAAM;EAC9D;CACF;CAEA,MAAM,YAAY,OAAwC;EACxD,IAAI;EACJ,IAAI;GACF,WAAW,MAAM,KAAK,YAAY,KAAK;EACzC,SAAS,OAAO;GACd,IAAK,MAA8B,WAAW,KAC5C,MAAM,IAAI,MAAM,wBAAwB,OAAO;GAEjD,MAAM;EACR;EAGA,IADe,KAAK,UAAU,SAAS,MACnC,MAAW,UACb,MAAM,IAAI,MACR,0BAA0B,SAAS,QAAQ,KAAK,SAAS,UAAU,GAAG,YAAY,OACpF;EAEF,MAAM,MAAM,SAAS,OAAO;EAC5B,IAAI,CAAC,KACH,MAAM,IAAI,MACR,gEAAgE,OAClE;EAGF,MAAM,QAAQ,oBAAoB,QAAQ;EAC1C,OAAO;GACL;GACA;GACA,GAAI,SAAS,EAAE,MAAM;EACvB;CACF;;;;;;CAOA,UACE,WACmD;EACnD,QAAQ,WAAR;GACE,KAAK;GACL,KAAK,UACH,OAAO;GACT,KAAK;GACL,KAAK;GACL,KAAK,aACH,OAAO;GACT,KAAK;GACL,KAAK;GACL,KAAK;GACL,KAAK,aACH,OAAO;GACT,KAAK,KAAA;GACL,SACE,OAAO;EACX;CACF;;;;;CAMA,qBAEE;EACA,OAAO,4BAA4B,KAAK,KAAK;CAC/C;;;;;CAMA,aACE,SACkD;EAClD,OAAO,qBAAqB,SAAS,KAAK,mBAAmB,CAAC;CAChE;AACF;;;;;;;;;;;;;;;;;;;;;;;;AAyBA,SAAgB,gBACd,OACA,QACA,QAC0B;CAC1B,OAAO,IAAI,iBAAiB;EAAE;EAAQ,GAAG;CAAO,GAAG,KAAK;AAC1D;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmCA,SAAgB,UACd,OACA,QAC0B;CAE1B,OAAO,gBAAgB,OADR,qBACe,GAAQ,MAAM;AAC9C"}
1
+ {"version":3,"file":"video.js","names":[],"sources":["../../../src/adapters/video.ts"],"sourcesContent":["import { resolveMediaPrompt } from '@tanstack/ai'\nimport { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'\nimport { toRunErrorPayload } from '@tanstack/ai/adapter-internals'\nimport { getGrokApiKeyFromEnv, withGrokDefaults } from '../utils/client'\nimport {\n GROK_VIDEO_MAX_REFERENCE_AUDIOS,\n GROK_VIDEO_MAX_REFERENCE_IMAGES,\n getGrokVideoDurationOptions,\n isGrokVideoReferenceModel,\n isGrokVideoSourceModel,\n parseGrokVideoSize,\n validateVideoSize,\n} from '../video/video-provider-options'\nimport type { DurationOptions } from '@tanstack/ai/adapters'\nimport type {\n ImagePart,\n MediaInputMetadata,\n TokenUsage,\n VideoGenerationOptions,\n VideoJobResult,\n VideoPart,\n VideoStatusResult,\n VideoUrlResult,\n} from '@tanstack/ai'\nimport type { GrokVideoModel } from '../model-meta'\nimport type {\n GrokVideoModelDurationByName,\n GrokVideoModelInputModalitiesByName,\n GrokVideoModelProviderOptionsByName,\n GrokVideoModelSizeByName,\n GrokVideoRuntimeOptions,\n} from '../video/video-provider-options'\nimport type { GrokClientConfig } from '../utils/client'\n\n/**\n * Configuration for Grok video adapter.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GrokVideoConfig extends GrokClientConfig {}\n\n/**\n * xAI bills video generation in \"USD ticks\": 10^10 ticks per US dollar\n * (e.g. one grok-imagine-video-1.5 second costs $0.08 = 800_000_000 ticks).\n */\nconst USD_TICKS_PER_DOLLAR = 10_000_000_000\n\n/** Response of the POST /v1/videos/{generations,edits,extensions} endpoints. */\ninterface GrokVideoCreateResponse {\n request_id?: string\n}\n\n/** Response of GET /v1/videos/{request_id}. */\ninterface GrokVideoStatusResponse {\n status?: string\n progress?: number\n model?: string\n video?: {\n url?: string\n duration?: number\n }\n usage?: {\n cost_in_usd_ticks?: number\n }\n error?: string\n}\n\n/**\n * Convert a TanStack image / video part to the URL string accepted by xAI's\n * Imagine video endpoints: public URLs pass through (fetched by xAI's\n * servers), data sources become base64 data URIs.\n */\nfunction mediaPartToUrl(\n part: ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata>,\n): string {\n if (part.source.type === 'url') return part.source.value\n return `data:${part.source.mimeType};base64,${part.source.value}`\n}\n\nfunction buildGrokVideoUsage(\n response: GrokVideoStatusResponse,\n): TokenUsage | undefined {\n const seconds = response.video?.duration\n const ticks = response.usage?.cost_in_usd_ticks\n if (seconds === undefined && ticks === undefined) return undefined\n return {\n promptTokens: 0,\n completionTokens: 0,\n totalTokens: 0,\n ...(seconds !== undefined && {\n billed: { quantity: seconds, unit: 'seconds' },\n unitsBilled: seconds,\n }),\n ...(ticks !== undefined && { cost: ticks / USD_TICKS_PER_DOLLAR }),\n }\n}\n\n/**\n * Grok Video Generation Adapter (xAI Imagine API)\n *\n * Tree-shakeable adapter for the grok-imagine video models using the\n * async jobs/polling architecture: create a generation request, poll it,\n * then read the completed video URL.\n *\n * Both models support text-to-video and image-to-video;\n * `grok-imagine-video-1.5` is xAI's documented default and adds native\n * 1080p generation plus reference-to-video inputs. Source-video edit\n * and extend are `grok-imagine-video` only.\n *\n * The Imagine video endpoints are not part of the OpenAI SDK surface (and\n * xAI rejects the SDK's multipart paths), so requests are plain JSON calls\n * issued with the configured `fetch` (or the global one).\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * Features:\n * - Async job-based video generation (1–15 second clips with audio)\n * - Aspect-ratio sizing via the \"aspectRatio_resolution\" size template\n * (e.g. '16:9_720p'), consistent with the grok-imagine image models\n * - Image-to-video via an `image` prompt part (starting frame URL or data URI)\n * - Reference-to-video via image prompt parts with\n * `metadata.role: 'reference'` or `'character'` (→ `reference_images`)\n * and preset voices via `modelOptions.reference_audios`\n * (grok-imagine-video-1.5 only)\n * - Video editing / extension on `grok-imagine-video` via a source\n * `video` prompt part and `modelOptions.mode: 'edit' | 'extend'`\n * (`/v1/videos/edits` / `/v1/videos/extensions`; in extend mode\n * `duration` is the added tail)\n * - Usage reporting: billed seconds (`usage.billed`) and exact cost\n */\nexport class GrokVideoAdapter<\n TModel extends GrokVideoModel,\n> extends BaseVideoAdapter<\n TModel,\n GrokVideoModelProviderOptionsByName[TModel],\n GrokVideoModelProviderOptionsByName,\n GrokVideoModelSizeByName,\n GrokVideoModelInputModalitiesByName,\n GrokVideoModelDurationByName\n> {\n readonly name = 'grok' as const\n\n private readonly clientConfig: GrokVideoConfig\n\n constructor(config: GrokVideoConfig, model: TModel) {\n super({}, model)\n this.clientConfig = withGrokDefaults(config)\n }\n\n private async request(\n path: string,\n init?: Omit<RequestInit, 'headers'>,\n ): Promise<Response> {\n // workerd's fetch must be invoked with this === globalThis\n const fetchFn = this.clientConfig.fetch ?? globalThis.fetch.bind(globalThis)\n return await fetchFn(`${this.clientConfig.baseURL}${path}`, {\n ...init,\n headers: {\n 'Content-Type': 'application/json',\n Authorization: `Bearer ${this.clientConfig.apiKey}`,\n },\n })\n }\n\n /**\n * Reads the error message out of an Imagine API error body\n * (`{\"code\": \"...\", \"error\": \"...\"}`), falling back to the raw text.\n */\n private async errorMessage(response: Response): Promise<string> {\n const body = await response.text()\n try {\n const parsed: unknown = JSON.parse(body)\n if (\n typeof parsed === 'object' &&\n parsed !== null &&\n 'error' in parsed &&\n typeof parsed.error === 'string'\n ) {\n return parsed.error\n }\n } catch {\n // not JSON — fall through to the raw body\n }\n return body\n }\n\n async createVideoJob(\n options: VideoGenerationOptions<\n GrokVideoModelProviderOptionsByName[TModel],\n GrokVideoModelSizeByName[TModel],\n GrokVideoModelDurationByName[TModel]\n >,\n ): Promise<VideoJobResult> {\n const { model, size, modelOptions, logger } = options\n\n // `mode` is a routing hint for this adapter, not an API field — strip it\n // before the remaining options are spread onto the request body. The\n // per-model map narrows what callers can pass, but modelOptions often\n // arrives as deserialized JSON, so the adapter handles the widest option\n // surface (the 1.5 shape) uniformly and gates by model at runtime.\n const { mode, ...wireOptions } = (modelOptions ??\n {}) as GrokVideoRuntimeOptions\n\n // `mode` is typed 'edit' | 'extend' but reaches us untrusted from JSON\n // callers. An unrecognised value must not fall through to the\n // generations endpoint with a source-video body — that would silently\n // run (and bill) a generation the caller never asked for.\n if (mode !== undefined && mode !== 'edit' && mode !== 'extend') {\n throw new Error(\n `${this.name}: unknown modelOptions.mode '${String(mode)}'. ` +\n `Expected 'edit' or 'extend'.`,\n )\n }\n\n // The interleaved prompt decomposes into verbatim text plus typed media\n // buckets. Reference audio is voice-id based (not an audio file), so\n // audio prompt parts have no request field to land in.\n const resolved = resolveMediaPrompt(options.prompt)\n if (resolved.audios.length > 0) {\n throw new Error(\n `${this.name}.createVideoJob does not support audio prompt parts (model: ${model}). ` +\n `To reference a preset voice, pass modelOptions.reference_audios ` +\n `(e.g. [{ voice_id: 'eve' }]).`,\n )\n }\n\n // A video prompt part is the source clip for edit / extension mode.\n // Those endpoints are grok-imagine-video only — 1.5 has no video input.\n if (\n !isGrokVideoSourceModel(model) &&\n (mode !== undefined || resolved.videos.length > 0)\n ) {\n throw new Error(\n `${this.name}: ${model} does not support video editing or extension. ` +\n `Use 'grok-imagine-video' for /v1/videos/edits and /v1/videos/extensions.`,\n )\n }\n\n // The mode must be chosen explicitly because the two endpoints have\n // different semantics (edit rewrites the clip, extend appends\n // `duration` seconds).\n if (resolved.videos.length > 1) {\n throw new Error(\n `${this.name}: ${model} accepts at most one source video; received ${resolved.videos.length}.`,\n )\n }\n const [sourceVideo] = resolved.videos\n if (sourceVideo && mode === undefined) {\n throw new Error(\n `${this.name}: a video prompt part needs modelOptions.mode set to ` +\n `'edit' (rewrite the clip) or 'extend' (append to it).`,\n )\n }\n if (!sourceVideo && mode !== undefined) {\n throw new Error(\n `${this.name}: modelOptions.mode '${mode}' requires a video prompt ` +\n `part carrying the source clip.`,\n )\n }\n\n if (mode !== undefined && sourceVideo) {\n return await this.createSourceVideoJob({\n model,\n mode,\n sourceVideo,\n resolved,\n wireOptions,\n size,\n genericDuration: options.duration,\n logger,\n })\n }\n\n validateVideoSize(model, size)\n\n // Pull the specially-handled keys out of the wire options: `duration`\n // is folded into the snapped value below, and the reference fields are\n // re-added explicitly so a JSON-serialized `null` or empty array reads\n // as \"unset\" instead of leaking onto the wire.\n const {\n duration: rawOptionDuration,\n reference_images: explicitReferenceImages,\n reference_audios: referenceAudios,\n ...generationOptions\n } = wireOptions\n\n // Coerce the requested duration into the model's valid range (1–15s,\n // integer) instead of rejecting it — `snapDuration` clamps and rounds.\n // modelOptions wins over the generic `duration`, mirroring the size\n // precedence below.\n const rawDuration = rawOptionDuration ?? options.duration\n const duration =\n rawDuration != null ? this.snapDuration(rawDuration) : undefined\n\n // Image parts split by role: un-roled / 'start_frame' images become the\n // starting frame (image-to-video); 'reference' / 'character' images\n // become reference_images (reference-to-video). The Imagine API has no\n // mask / control / end-frame inputs. Unknown role strings (possible via\n // JSON callers) throw rather than silently dropping the part.\n const startFrames: Array<ImagePart<MediaInputMetadata>> = []\n const referenceImages: Array<{ url: string }> = []\n for (const part of resolved.images) {\n const role = part.metadata?.role\n switch (role) {\n case 'mask':\n case 'control':\n case 'end_frame':\n throw new Error(\n `${this.name}: the Imagine video API has no '${role}' image ` +\n `input on model ${model}. Use an un-roled / 'start_frame' ` +\n `image as the starting frame, or 'reference' images.`,\n )\n case 'reference':\n case 'character':\n referenceImages.push({ url: mediaPartToUrl(part) })\n break\n case 'start_frame':\n case undefined:\n startFrames.push(part)\n break\n default:\n throw new Error(\n `${this.name}: unknown image metadata.role '${String(role)}'. ` +\n `Expected 'start_frame', 'reference', or 'character'.`,\n )\n }\n }\n if (startFrames.length > 1) {\n throw new Error(\n `${this.name}: ${model} accepts at most one starting-frame image; received ${startFrames.length}. ` +\n `Use metadata.role: 'reference' for reference-to-video inputs.`,\n )\n }\n // Explicit modelOptions.reference_images replaces the part-derived list\n // (an explicit empty array means \"none\").\n const finalReferenceImages =\n explicitReferenceImages ??\n (referenceImages.length > 0 ? referenceImages : undefined)\n const referenceImageCount = finalReferenceImages?.length ?? 0\n const referenceAudioCount = referenceAudios?.length ?? 0\n const hasReference = referenceImageCount > 0 || referenceAudioCount > 0\n\n // Reference inputs are a grok-imagine-video-1.5 feature. The per-model\n // options map already hides the fields from other models at compile\n // time; this runtime gate covers prompt-part roles and untyped callers.\n if (!isGrokVideoReferenceModel(model) && hasReference) {\n throw new Error(\n `${this.name}: ${model} does not support reference-to-video inputs. ` +\n `Use 'grok-imagine-video-1.5' for reference_images / reference_audios.`,\n )\n }\n if (referenceAudioCount > GROK_VIDEO_MAX_REFERENCE_AUDIOS) {\n throw new Error(\n `${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_AUDIOS} reference voices; received ${referenceAudioCount}.`,\n )\n }\n if (referenceImageCount > GROK_VIDEO_MAX_REFERENCE_IMAGES) {\n throw new Error(\n `${this.name}: ${model} accepts at most ${GROK_VIDEO_MAX_REFERENCE_IMAGES} reference images; received ${referenceImageCount}.`,\n )\n }\n\n // Image-to-video: the single image prompt part becomes the starting frame\n // and the prompt text describes the desired motion. URL sources are\n // fetched by xAI's servers; data sources are sent as base64 data URIs.\n // On grok-imagine-video-1.5 the starting frame combines with reference\n // inputs — `image` pins the first frame while reference_images /\n // reference_audios steer subjects and voices. Classic grok-imagine-video\n // rejects that mix, but it rejects reference inputs outright, so the\n // model gate above already covers it.\n const [startFrame] = startFrames\n\n // The generic `size` option carries an \"aspectRatio_resolution\" template\n // (e.g. '16:9_720p') and maps to the Imagine API's `aspect_ratio` /\n // `resolution` parameters; explicit modelOptions win over the template\n // (including `reference_images`, which replaces the part-derived list).\n const parsedSize = size !== undefined ? parseGrokVideoSize(size) : undefined\n const resolvedResolution =\n generationOptions.resolution ?? parsedSize?.resolution\n if (hasReference && resolvedResolution === '1080p') {\n throw new Error(\n `${this.name}: reference-to-video is capped at 720p on ${model}.`,\n )\n }\n const request = {\n model,\n prompt: resolved.text,\n ...(startFrame && { image: { url: mediaPartToUrl(startFrame) } }),\n ...(referenceImageCount > 0 && {\n reference_images: finalReferenceImages,\n }),\n ...(referenceAudioCount > 0 && {\n reference_audios: referenceAudios,\n }),\n ...(parsedSize && {\n aspect_ratio: parsedSize.aspectRatio,\n ...(parsedSize.resolution !== undefined && {\n resolution: parsedSize.resolution,\n }),\n }),\n // The remaining options spread after the size template so explicit\n // aspect_ratio / resolution win over it; duration and the reference\n // fields were destructured out above and re-added normalized.\n ...generationOptions,\n ...(duration !== undefined && { duration }),\n }\n\n return await this.postVideoJob('/videos/generations', request, {\n model,\n logger,\n logLine: `activity=video.create provider=${this.name} model=${model} mode=generate size=${size ?? 'default'} duration=${duration ?? 'default'}`,\n })\n }\n\n /**\n * Build and post an edit / extension request. Both endpoints take only\n * `model`, `prompt`, and the source `video` (plus `duration` — the length\n * of the added tail — for extensions): output geometry is inherited from\n * the source clip, capped at 720p, and edit outputs also inherit the\n * source length. Rather than sending fields the API documents as ignored,\n * the inapplicable options are rejected with actionable errors.\n */\n private async createSourceVideoJob(args: {\n model: string\n mode: 'edit' | 'extend'\n sourceVideo: VideoPart<MediaInputMetadata>\n resolved: ReturnType<typeof resolveMediaPrompt>\n wireOptions: Omit<GrokVideoRuntimeOptions, 'mode'>\n size: string | undefined\n genericDuration: number | undefined\n logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']\n }): Promise<VideoJobResult> {\n const { model, mode, sourceVideo, resolved, wireOptions, logger } = args\n const endpoint = mode === 'edit' ? '/videos/edits' : '/videos/extensions'\n\n if (resolved.images.length > 0) {\n throw new Error(\n `${this.name}: '${mode}' mode takes only the source video — image ` +\n `prompt parts are not supported by ${endpoint}.`,\n )\n }\n\n // Pull every generation-only key out of the wire options so nothing can\n // leak into the edit/extend body via the spread below. JSON-serialized\n // `null` values (a common \"unset\" encoding) are treated as absent;\n // actual values are rejected with actionable errors.\n const {\n aspect_ratio: aspectRatio,\n resolution,\n duration: modeDuration,\n reference_images: referenceImagesOption,\n reference_audios: referenceAudiosOption,\n ...passthrough\n } = wireOptions\n if (\n (referenceImagesOption?.length ?? 0) > 0 ||\n (referenceAudiosOption?.length ?? 0) > 0\n ) {\n throw new Error(\n `${this.name}: reference inputs are only supported by video ` +\n `generation, not '${mode}' mode.`,\n )\n }\n if (args.size !== undefined || aspectRatio != null || resolution != null) {\n throw new Error(\n `${this.name}: '${mode}' mode does not accept size / aspect_ratio / ` +\n `resolution — the output inherits the source clip's geometry ` +\n `(capped at 720p).`,\n )\n }\n const rawDuration = modeDuration ?? args.genericDuration\n if (mode === 'edit' && rawDuration != null) {\n throw new Error(\n `${this.name}: 'edit' mode does not accept a duration — the output ` +\n `inherits the source clip's length. Use mode 'extend' to append ` +\n `seconds to the clip.`,\n )\n }\n // Extend: the snapped duration is the added-tail length (1–15s).\n const duration =\n rawDuration != null ? this.snapDuration(rawDuration) : undefined\n\n const request = {\n model,\n prompt: resolved.text,\n video: { url: mediaPartToUrl(sourceVideo) },\n ...passthrough,\n ...(duration !== undefined && { duration }),\n }\n\n return await this.postVideoJob(endpoint, request, {\n model,\n logger,\n logLine: `activity=video.create provider=${this.name} model=${model} mode=${mode} duration=${duration ?? 'default'}`,\n })\n }\n\n /**\n * POST a create-job request body to one of the Imagine video endpoints\n * (`/videos/generations`, `/videos/edits`, `/videos/extensions`) and read\n * the `request_id` out of the shared response shape.\n */\n private async postVideoJob(\n endpoint: string,\n request: Record<string, unknown>,\n context: {\n model: string\n logger: VideoGenerationOptions<GrokVideoRuntimeOptions>['logger']\n logLine: string\n },\n ): Promise<VideoJobResult> {\n const { model, logger, logLine } = context\n try {\n logger.request(logLine, { provider: this.name, model })\n\n const response = await this.request(endpoint, {\n method: 'POST',\n body: JSON.stringify(request),\n })\n if (!response.ok) {\n throw new Error(\n `grok: ${endpoint} request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,\n )\n }\n\n const result = (await response.json()) as GrokVideoCreateResponse\n if (!result.request_id) {\n throw new Error(`grok: ${endpoint} response contained no request_id`)\n }\n return { jobId: result.request_id, model }\n } catch (error: unknown) {\n logger.errors(`${this.name}.createVideoJob fatal`, {\n error: toRunErrorPayload(error, `${this.name}.createVideoJob failed`),\n source: `${this.name}.createVideoJob`,\n })\n throw error\n }\n }\n\n private async retrieveJob(jobId: string): Promise<GrokVideoStatusResponse> {\n const response = await this.request(`/videos/${jobId}`)\n if (!response.ok) {\n const error = new Error(\n `grok: video status request failed (${response.status} ${response.statusText}): ${await this.errorMessage(response)}`,\n )\n ;(error as { status?: number }).status = response.status\n throw error\n }\n return (await response.json()) as GrokVideoStatusResponse\n }\n\n async getVideoStatus(jobId: string): Promise<VideoStatusResult> {\n let response: GrokVideoStatusResponse\n try {\n response = await this.retrieveJob(jobId)\n } catch (error) {\n if ((error as { status?: number }).status === 404) {\n return { jobId, status: 'failed', error: 'Job not found' }\n }\n throw error\n }\n\n return {\n jobId,\n status: this.mapStatus(response.status),\n ...(response.progress !== undefined && { progress: response.progress }),\n ...(response.error !== undefined && { error: response.error }),\n }\n }\n\n async getVideoUrl(jobId: string): Promise<VideoUrlResult> {\n let response: GrokVideoStatusResponse\n try {\n response = await this.retrieveJob(jobId)\n } catch (error) {\n if ((error as { status?: number }).status === 404) {\n throw new Error(`Video job not found: ${jobId}`)\n }\n throw error\n }\n\n const status = this.mapStatus(response.status)\n if (status === 'failed') {\n throw new Error(\n `Video generation failed${response.error ? `: ${response.error}` : ''}. Job ID: ${jobId}`,\n )\n }\n const url = response.video?.url\n if (!url) {\n throw new Error(\n `Video is not ready for download. Check status first. Job ID: ${jobId}`,\n )\n }\n\n const usage = buildGrokVideoUsage(response)\n return {\n jobId,\n url,\n ...(usage && { usage }),\n }\n }\n\n /**\n * Maps Imagine API job statuses onto the generic video status set. The\n * API reports 'pending' while queued/generating (with a numeric\n * `progress`), then a terminal 'done' / 'failed' / 'expired'.\n */\n protected mapStatus(\n apiStatus: string | undefined,\n ): 'pending' | 'processing' | 'completed' | 'failed' {\n switch (apiStatus) {\n case 'pending':\n case 'queued':\n return 'pending'\n case 'done':\n case 'completed':\n case 'succeeded':\n return 'completed'\n case 'failed':\n case 'expired':\n case 'error':\n case 'cancelled':\n return 'failed'\n case undefined:\n default:\n return 'processing'\n }\n }\n\n /**\n * Both grok-imagine video models accept a continuous 1–15 integer-second\n * range. Consumers can use this to render UI without provider knowledge.\n */\n override availableDurations(): DurationOptions<\n GrokVideoModelDurationByName[TModel]\n > {\n return getGrokVideoDurationOptions(this.model)\n }\n\n /**\n * Coerce a raw seconds value to the closest valid duration (clamped to\n * [1, 15] and rounded to whole seconds).\n */\n override snapDuration(\n seconds: number,\n ): GrokVideoModelDurationByName[TModel] | undefined {\n return snapToDurationOption(seconds, this.availableDurations())\n }\n}\n\n/**\n * Creates a Grok video adapter with an explicit API key.\n * Type resolution happens here at the call site.\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'grok-imagine-video-1.5')\n * @param apiKey - Your xAI API key\n * @param config - Optional additional configuration\n * @returns Configured Grok video adapter instance with resolved types\n *\n * @example\n * ```typescript\n * const adapter = createGrokVideo('grok-imagine-video-1.5', 'xai-...');\n *\n * const { jobId } = await generateVideo({\n * adapter,\n * prompt: 'A beautiful sunset over the ocean',\n * size: '16:9_720p',\n * duration: 5\n * });\n * ```\n */\nexport function createGrokVideo<TModel extends GrokVideoModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GrokVideoConfig, 'apiKey'>,\n): GrokVideoAdapter<TModel> {\n return new GrokVideoAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Grok video adapter with automatic API key detection from environment variables.\n * Type resolution happens here at the call site.\n *\n * Looks for `XAI_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (Browser with injected env)\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'grok-imagine-video-1.5')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Grok video adapter instance with resolved types\n * @throws Error if XAI_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * // Automatically uses XAI_API_KEY from environment\n * const adapter = grokVideo('grok-imagine-video-1.5');\n *\n * // Image-to-video: an optional image prompt part is the starting frame.\n * const { jobId } = await generateVideo({\n * adapter,\n * prompt: [\n * { type: 'text', content: 'Make the cat start playing the piano' },\n * { type: 'image', source: { type: 'url', value: 'https://example.com/cat.png' } },\n * ],\n * });\n *\n * // Poll for status\n * const status = await getVideoJobStatus({ adapter, jobId });\n * ```\n */\nexport function grokVideo<TModel extends GrokVideoModel>(\n model: TModel,\n config?: Omit<GrokVideoConfig, 'apiKey'>,\n): GrokVideoAdapter<TModel> {\n const apiKey = getGrokApiKeyFromEnv()\n return createGrokVideo(model, apiKey, config)\n}\n"],"mappings":";;;;;;;;;;AA6CA,IAAM,uBAAuB;;;;;;AA2B7B,SAAS,eACP,MACQ;CACR,IAAI,KAAK,OAAO,SAAS,OAAO,OAAO,KAAK,OAAO;CACnD,OAAO,QAAQ,KAAK,OAAO,SAAS,UAAU,KAAK,OAAO;AAC5D;AAEA,SAAS,oBACP,UACwB;CACxB,MAAM,UAAU,SAAS,OAAO;CAChC,MAAM,QAAQ,SAAS,OAAO;CAC9B,IAAI,YAAY,KAAA,KAAa,UAAU,KAAA,GAAW,OAAO,KAAA;CACzD,OAAO;EACL,cAAc;EACd,kBAAkB;EAClB,aAAa;EACb,GAAI,YAAY,KAAA,KAAa;GAC3B,QAAQ;IAAE,UAAU;IAAS,MAAM;GAAU;GAC7C,aAAa;EACf;EACA,GAAI,UAAU,KAAA,KAAa,EAAE,MAAM,QAAQ,qBAAqB;CAClE;AACF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmCA,IAAa,mBAAb,cAEU,iBAOR;CACA,OAAgB;CAEhB;CAEA,YAAY,QAAyB,OAAe;EAClD,MAAM,CAAC,GAAG,KAAK;EACf,KAAK,eAAe,iBAAiB,MAAM;CAC7C;CAEA,MAAc,QACZ,MACA,MACmB;EAGnB,OAAO,OADS,KAAK,aAAa,SAAS,WAAW,MAAM,KAAK,UAAU,EAAA,CACtD,GAAG,KAAK,aAAa,UAAU,QAAQ;GAC1D,GAAG;GACH,SAAS;IACP,gBAAgB;IAChB,eAAe,UAAU,KAAK,aAAa;GAC7C;EACF,CAAC;CACH;;;;;CAMA,MAAc,aAAa,UAAqC;EAC9D,MAAM,OAAO,MAAM,SAAS,KAAK;EACjC,IAAI;GACF,MAAM,SAAkB,KAAK,MAAM,IAAI;GACvC,IACE,OAAO,WAAW,YAClB,WAAW,QACX,WAAW,UACX,OAAO,OAAO,UAAU,UAExB,OAAO,OAAO;EAElB,QAAQ,CAER;EACA,OAAO;CACT;CAEA,MAAM,eACJ,SAKyB;EACzB,MAAM,EAAE,OAAO,MAAM,cAAc,WAAW;EAO9C,MAAM,EAAE,MAAM,GAAG,gBAAiB,gBAChC,CAAC;EAMH,IAAI,SAAS,KAAA,KAAa,SAAS,UAAU,SAAS,UACpD,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,+BAA+B,OAAO,IAAI,EAAE,gCAE3D;EAMF,MAAM,WAAW,mBAAmB,QAAQ,MAAM;EAClD,IAAI,SAAS,OAAO,SAAS,GAC3B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,8DAA8D,MAAM,iGAGnF;EAKF,IACE,CAAC,uBAAuB,KAAK,MAC5B,SAAS,KAAA,KAAa,SAAS,OAAO,SAAS,IAEhD,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,uHAEzB;EAMF,IAAI,SAAS,OAAO,SAAS,GAC3B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,8CAA8C,SAAS,OAAO,OAAO,EAC9F;EAEF,MAAM,CAAC,eAAe,SAAS;EAC/B,IAAI,eAAe,SAAS,KAAA,GAC1B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,2GAEf;EAEF,IAAI,CAAC,eAAe,SAAS,KAAA,GAC3B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,uBAAuB,KAAK,yDAE3C;EAGF,IAAI,SAAS,KAAA,KAAa,aACxB,OAAO,MAAM,KAAK,qBAAqB;GACrC;GACA;GACA;GACA;GACA;GACA;GACA,iBAAiB,QAAQ;GACzB;EACF,CAAC;EAGH,kBAAkB,OAAO,IAAI;EAM7B,MAAM,EACJ,UAAU,mBACV,kBAAkB,yBAClB,kBAAkB,iBAClB,GAAG,sBACD;EAMJ,MAAM,cAAc,qBAAqB,QAAQ;EACjD,MAAM,WACJ,eAAe,OAAO,KAAK,aAAa,WAAW,IAAI,KAAA;EAOzD,MAAM,cAAoD,CAAC;EAC3D,MAAM,kBAA0C,CAAC;EACjD,KAAK,MAAM,QAAQ,SAAS,QAAQ;GAClC,MAAM,OAAO,KAAK,UAAU;GAC5B,QAAQ,MAAR;IACE,KAAK;IACL,KAAK;IACL,KAAK,aACH,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,kCAAkC,KAAK,yBAChC,MAAM,sFAE5B;IACF,KAAK;IACL,KAAK;KACH,gBAAgB,KAAK,EAAE,KAAK,eAAe,IAAI,EAAE,CAAC;KAClD;IACF,KAAK;IACL,KAAK,KAAA;KACH,YAAY,KAAK,IAAI;KACrB;IACF,SACE,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,iCAAiC,OAAO,IAAI,EAAE,wDAE7D;GACJ;EACF;EACA,IAAI,YAAY,SAAS,GACvB,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,sDAAsD,YAAY,OAAO,gEAElG;EAIF,MAAM,uBACJ,4BACC,gBAAgB,SAAS,IAAI,kBAAkB,KAAA;EAClD,MAAM,sBAAsB,sBAAsB,UAAU;EAC5D,MAAM,sBAAsB,iBAAiB,UAAU;EACvD,MAAM,eAAe,sBAAsB,KAAK,sBAAsB;EAKtE,IAAI,CAAC,0BAA0B,KAAK,KAAK,cACvC,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,mHAEzB;EAEF,IAAI,sBAAA,GACF,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,gDAAiF,oBAAoB,EAC9H;EAEF,IAAI,sBAAA,GACF,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,IAAI,MAAM,gDAAiF,oBAAoB,EAC9H;EAWF,MAAM,CAAC,cAAc;EAMrB,MAAM,aAAa,SAAS,KAAA,IAAY,mBAAmB,IAAI,IAAI,KAAA;EACnE,MAAM,qBACJ,kBAAkB,cAAc,YAAY;EAC9C,IAAI,gBAAgB,uBAAuB,SACzC,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,4CAA4C,MAAM,EACjE;EAEF,MAAM,UAAU;GACd;GACA,QAAQ,SAAS;GACjB,GAAI,cAAc,EAAE,OAAO,EAAE,KAAK,eAAe,UAAU,EAAE,EAAE;GAC/D,GAAI,sBAAsB,KAAK,EAC7B,kBAAkB,qBACpB;GACA,GAAI,sBAAsB,KAAK,EAC7B,kBAAkB,gBACpB;GACA,GAAI,cAAc;IAChB,cAAc,WAAW;IACzB,GAAI,WAAW,eAAe,KAAA,KAAa,EACzC,YAAY,WAAW,WACzB;GACF;GAIA,GAAG;GACH,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;EAC3C;EAEA,OAAO,MAAM,KAAK,aAAa,uBAAuB,SAAS;GAC7D;GACA;GACA,SAAS,kCAAkC,KAAK,KAAK,SAAS,MAAM,sBAAsB,QAAQ,UAAU,YAAY,YAAY;EACtI,CAAC;CACH;;;;;;;;;CAUA,MAAc,qBAAqB,MASP;EAC1B,MAAM,EAAE,OAAO,MAAM,aAAa,UAAU,aAAa,WAAW;EACpE,MAAM,WAAW,SAAS,SAAS,kBAAkB;EAErD,IAAI,SAAS,OAAO,SAAS,GAC3B,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,KAAK,KAAK,+EACgB,SAAS,EAClD;EAOF,MAAM,EACJ,cAAc,aACd,YACA,UAAU,cACV,kBAAkB,uBAClB,kBAAkB,uBAClB,GAAG,gBACD;EACJ,KACG,uBAAuB,UAAU,KAAK,MACtC,uBAAuB,UAAU,KAAK,GAEvC,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,kEACS,KAAK,QAC7B;EAEF,IAAI,KAAK,SAAS,KAAA,KAAa,eAAe,QAAQ,cAAc,MAClE,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,KAAK,KAAK,2HAGzB;EAEF,MAAM,cAAc,gBAAgB,KAAK;EACzC,IAAI,SAAS,UAAU,eAAe,MACpC,MAAM,IAAI,MACR,GAAG,KAAK,KAAK,0IAGf;EAGF,MAAM,WACJ,eAAe,OAAO,KAAK,aAAa,WAAW,IAAI,KAAA;EAEzD,MAAM,UAAU;GACd;GACA,QAAQ,SAAS;GACjB,OAAO,EAAE,KAAK,eAAe,WAAW,EAAE;GAC1C,GAAG;GACH,GAAI,aAAa,KAAA,KAAa,EAAE,SAAS;EAC3C;EAEA,OAAO,MAAM,KAAK,aAAa,UAAU,SAAS;GAChD;GACA;GACA,SAAS,kCAAkC,KAAK,KAAK,SAAS,MAAM,QAAQ,KAAK,YAAY,YAAY;EAC3G,CAAC;CACH;;;;;;CAOA,MAAc,aACZ,UACA,SACA,SAKyB;EACzB,MAAM,EAAE,OAAO,QAAQ,YAAY;EACnC,IAAI;GACF,OAAO,QAAQ,SAAS;IAAE,UAAU,KAAK;IAAM;GAAM,CAAC;GAEtD,MAAM,WAAW,MAAM,KAAK,QAAQ,UAAU;IAC5C,QAAQ;IACR,MAAM,KAAK,UAAU,OAAO;GAC9B,CAAC;GACD,IAAI,CAAC,SAAS,IACZ,MAAM,IAAI,MACR,SAAS,SAAS,mBAAmB,SAAS,OAAO,GAAG,SAAS,WAAW,KAAK,MAAM,KAAK,aAAa,QAAQ,GACnH;GAGF,MAAM,SAAU,MAAM,SAAS,KAAK;GACpC,IAAI,CAAC,OAAO,YACV,MAAM,IAAI,MAAM,SAAS,SAAS,kCAAkC;GAEtE,OAAO;IAAE,OAAO,OAAO;IAAY;GAAM;EAC3C,SAAS,OAAgB;GACvB,OAAO,OAAO,GAAG,KAAK,KAAK,wBAAwB;IACjD,OAAO,kBAAkB,OAAO,GAAG,KAAK,KAAK,uBAAuB;IACpE,QAAQ,GAAG,KAAK,KAAK;GACvB,CAAC;GACD,MAAM;EACR;CACF;CAEA,MAAc,YAAY,OAAiD;EACzE,MAAM,WAAW,MAAM,KAAK,QAAQ,WAAW,OAAO;EACtD,IAAI,CAAC,SAAS,IAAI;GAChB,MAAM,wBAAQ,IAAI,MAChB,sCAAsC,SAAS,OAAO,GAAG,SAAS,WAAW,KAAK,MAAM,KAAK,aAAa,QAAQ,GACpH;GACC,MAA+B,SAAS,SAAS;GAClD,MAAM;EACR;EACA,OAAQ,MAAM,SAAS,KAAK;CAC9B;CAEA,MAAM,eAAe,OAA2C;EAC9D,IAAI;EACJ,IAAI;GACF,WAAW,MAAM,KAAK,YAAY,KAAK;EACzC,SAAS,OAAO;GACd,IAAK,MAA8B,WAAW,KAC5C,OAAO;IAAE;IAAO,QAAQ;IAAU,OAAO;GAAgB;GAE3D,MAAM;EACR;EAEA,OAAO;GACL;GACA,QAAQ,KAAK,UAAU,SAAS,MAAM;GACtC,GAAI,SAAS,aAAa,KAAA,KAAa,EAAE,UAAU,SAAS,SAAS;GACrE,GAAI,SAAS,UAAU,KAAA,KAAa,EAAE,OAAO,SAAS,MAAM;EAC9D;CACF;CAEA,MAAM,YAAY,OAAwC;EACxD,IAAI;EACJ,IAAI;GACF,WAAW,MAAM,KAAK,YAAY,KAAK;EACzC,SAAS,OAAO;GACd,IAAK,MAA8B,WAAW,KAC5C,MAAM,IAAI,MAAM,wBAAwB,OAAO;GAEjD,MAAM;EACR;EAGA,IADe,KAAK,UAAU,SAAS,MACnC,MAAW,UACb,MAAM,IAAI,MACR,0BAA0B,SAAS,QAAQ,KAAK,SAAS,UAAU,GAAG,YAAY,OACpF;EAEF,MAAM,MAAM,SAAS,OAAO;EAC5B,IAAI,CAAC,KACH,MAAM,IAAI,MACR,gEAAgE,OAClE;EAGF,MAAM,QAAQ,oBAAoB,QAAQ;EAC1C,OAAO;GACL;GACA;GACA,GAAI,SAAS,EAAE,MAAM;EACvB;CACF;;;;;;CAOA,UACE,WACmD;EACnD,QAAQ,WAAR;GACE,KAAK;GACL,KAAK,UACH,OAAO;GACT,KAAK;GACL,KAAK;GACL,KAAK,aACH,OAAO;GACT,KAAK;GACL,KAAK;GACL,KAAK;GACL,KAAK,aACH,OAAO;GACT,KAAK,KAAA;GACL,SACE,OAAO;EACX;CACF;;;;;CAMA,qBAEE;EACA,OAAO,4BAA4B,KAAK,KAAK;CAC/C;;;;;CAMA,aACE,SACkD;EAClD,OAAO,qBAAqB,SAAS,KAAK,mBAAmB,CAAC;CAChE;AACF;;;;;;;;;;;;;;;;;;;;;;;;AAyBA,SAAgB,gBACd,OACA,QACA,QAC0B;CAC1B,OAAO,IAAI,iBAAiB;EAAE;EAAQ,GAAG;CAAO,GAAG,KAAK;AAC1D;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;AAmCA,SAAgB,UACd,OACA,QAC0B;CAE1B,OAAO,gBAAgB,OADR,qBACe,GAAQ,MAAM;AAC9C"}
@@ -166,7 +166,8 @@ export interface GrokVideoProviderOptions extends GrokVideoBaseProviderOptions {
166
166
  * `metadata.role: 'reference'` (or `'character'`); set explicitly to
167
167
  * replace the part-derived list. Reference images are addressed from the
168
168
  * prompt text as `<IMAGE_0>`, `<IMAGE_1>`, … in request order, and do not
169
- * lock the first frame.
169
+ * lock the first frame — pair them with a starting-frame image prompt part
170
+ * when the opening frame has to be pinned.
170
171
  */
171
172
  reference_images?: Array<{
172
173
  url: string;
@@ -234,6 +235,8 @@ export type GrokVideoModelSizeByName = {
234
235
  * Both models support text-to-video and accept an optional `image` prompt
235
236
  * part as the starting frame; image parts with `metadata.role: 'reference'`
236
237
  * or `'character'` become `reference_images` (grok-imagine-video-1.5 only).
238
+ * On grok-imagine-video-1.5 the two combine: the starting frame pins the
239
+ * first frame while the reference images steer subjects and style.
237
240
  * A `video` prompt part carries the source clip for edit / extension mode
238
241
  * on grok-imagine-video only (`modelOptions.mode: 'edit' | 'extend'`).
239
242
  *
@@ -1 +1 @@
1
- {"version":3,"file":"video-provider-options.js","names":[],"sources":["../../../src/video/video-provider-options.ts"],"sourcesContent":["/**\n * Grok Video Generation Provider Options (xAI Imagine API)\n *\n * Based on https://docs.x.ai/developers/model-capabilities/video/generation\n * (plus the image-to-video, reference-to-video, editing, and extension pages\n * under the same section).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\n\nimport type { DurationOptions } from '@tanstack/ai/adapters'\nimport type { GrokVideoModel } from '../model-meta'\n\n/**\n * Aspect ratios accepted by the grok-imagine video models.\n *\n * Note: this is a narrower set than the grok-imagine image models — the\n * video endpoint rejects the phone-screen ratios ('9:19.5', '9:20', …) and\n * 'auto'.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoAspectRatio =\n | '1:1'\n | '16:9'\n | '9:16'\n | '4:3'\n | '3:4'\n | '3:2'\n | '2:3'\n\n/**\n * Resolution tiers for the grok-imagine video models.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoResolution = '480p' | '720p' | '1080p'\n\n/**\n * Resolutions accepted by grok-imagine-video (v1.0). Native 1080p is a\n * grok-imagine-video-1.5 generation feature.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoResolutionV1 = '480p' | '720p'\n\n/**\n * Size strings for grok-imagine video models. The Imagine API is\n * aspect-ratio based rather than pixel-size based; like the grok-imagine\n * image models, the generic `size` option uses an\n * `aspectRatio_resolution` template (\"16:9_720p\") — the resolution suffix\n * is optional (\"16:9\" uses the API default).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoSize =\n | GrokVideoAspectRatio\n | `${GrokVideoAspectRatio}_${GrokVideoResolution}`\n\n/**\n * Size strings for grok-imagine-video (v1.0) — 1080p is not in the type.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoSizeV1 =\n | GrokVideoAspectRatio\n | `${GrokVideoAspectRatio}_${GrokVideoResolutionV1}`\n\nconst GROK_VIDEO_ASPECT_RATIOS: ReadonlyArray<string> = [\n '1:1',\n '16:9',\n '9:16',\n '4:3',\n '3:4',\n '3:2',\n '2:3',\n]\n\nconst GROK_VIDEO_RESOLUTIONS: ReadonlyArray<string> = ['480p', '720p', '1080p']\n\n/**\n * Video duration limits enforced by the Imagine API (seconds).\n */\nexport const GROK_VIDEO_MIN_DURATION = 1\nexport const GROK_VIDEO_MAX_DURATION = 15\n\n/**\n * Parses a grok video size string into its components.\n * Format: \"aspectRatio\" or \"aspectRatio_resolution\",\n * e.g. \"16:9_720p\" → { aspectRatio: \"16:9\", resolution: \"720p\" }.\n * Returns undefined when the string doesn't match the template.\n */\nexport function parseGrokVideoSize(\n size: string,\n): { aspectRatio: string; resolution?: string } | undefined {\n const match = size.match(/^([\\d.]+:[\\d.]+)(?:_(.+))?$/)\n const [, aspectRatio, resolution] = match ?? []\n if (aspectRatio === undefined) return undefined\n return { aspectRatio, ...(resolution !== undefined && { resolution }) }\n}\n\n/**\n * Models that accept native 1080p on text-to-video and image-to-video.\n * Reference-to-video stays capped at 720p even on these models.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function isGrokVideoNative1080pModel(model: string): boolean {\n return model === 'grok-imagine-video-1.5'\n}\n\n/**\n * Validate the `size` template for a given grok video model.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function validateVideoSize(\n model: string,\n size?: string,\n): asserts size is GrokVideoSize | undefined {\n if (size === undefined) return\n const parsed = parseGrokVideoSize(size)\n if (!parsed || !GROK_VIDEO_ASPECT_RATIOS.includes(parsed.aspectRatio)) {\n throw new Error(\n `Size \"${size}\" is not supported by model \"${model}\". Expected ` +\n `\"aspectRatio\" or \"aspectRatio_resolution\" (e.g. \"16:9_720p\") with ` +\n `aspect ratio one of: ${GROK_VIDEO_ASPECT_RATIOS.join(', ')}`,\n )\n }\n if (\n parsed.resolution !== undefined &&\n !GROK_VIDEO_RESOLUTIONS.includes(parsed.resolution)\n ) {\n throw new Error(\n `Resolution \"${parsed.resolution}\" is not supported by model \"${model}\". ` +\n `Supported resolutions: ${GROK_VIDEO_RESOLUTIONS.join(', ')}`,\n )\n }\n if (parsed.resolution === '1080p' && !isGrokVideoNative1080pModel(model)) {\n throw new Error(\n `Resolution \"1080p\" is not supported by model \"${model}\". ` +\n `Use 'grok-imagine-video-1.5' for native 1080p text-to-video / image-to-video.`,\n )\n }\n}\n\n/**\n * Per-model duration type. The Imagine API accepts any integer second in the\n * 1–15 range, so this is a continuous range expressed as `number` (a literal\n * union can't represent it). `snapDuration()` coerces a raw seconds value into\n * the valid range at runtime.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoModelDurationByName = {\n 'grok-imagine-video': number\n 'grok-imagine-video-1.5': number\n}\n\n/**\n * Runtime duration table backing `availableDurations()` / `snapDuration()`.\n * Both grok-imagine video models accept the same continuous 1–15 integer-second\n * range.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport const GROK_VIDEO_DURATIONS: {\n readonly [TModel in GrokVideoModel]: DurationOptions<\n GrokVideoModelDurationByName[TModel]\n >\n} = {\n 'grok-imagine-video': {\n kind: 'range',\n min: GROK_VIDEO_MIN_DURATION,\n max: GROK_VIDEO_MAX_DURATION,\n step: 1,\n unit: 'seconds',\n },\n 'grok-imagine-video-1.5': {\n kind: 'range',\n min: GROK_VIDEO_MIN_DURATION,\n max: GROK_VIDEO_MAX_DURATION,\n step: 1,\n unit: 'seconds',\n },\n}\n\n/**\n * Look up the duration options for a grok video model.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function getGrokVideoDurationOptions<TModel extends GrokVideoModel>(\n model: TModel,\n): DurationOptions<GrokVideoModelDurationByName[TModel]> {\n return GROK_VIDEO_DURATIONS[model]\n}\n\n/**\n * Request mode for a source-video job. `'edit'` posts to `/v1/videos/edits`\n * (modify the source clip in place); `'extend'` posts to\n * `/v1/videos/extensions` (continue the source clip — `duration` is the\n * length of the **added tail**, not the total). Both require exactly one\n * video prompt part carrying the source clip, and both are\n * `grok-imagine-video` (v1.0) only.\n *\n * Output geometry (aspect ratio / resolution) is inherited from the source\n * clip in both modes, capped at 720p, and edit outputs also inherit the\n * source length — the adapter rejects `size`, `aspect_ratio`, `resolution`,\n * and (in edit mode) `duration` rather than sending fields the API ignores.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoMode = 'edit' | 'extend'\n\n/**\n * Provider options shared by both grok-imagine video models. These map\n * directly onto the Imagine API request body and take precedence over the\n * generic `size` / `duration` options when both are provided.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GrokVideoBaseProviderOptions {\n /**\n * Output aspect ratio. Generation only — edit / extend outputs inherit\n * the source clip's geometry and the adapter rejects this in those modes.\n */\n aspect_ratio?: GrokVideoAspectRatio\n\n /**\n * Output resolution tier. Generation only — edit / extend outputs inherit\n * the source clip's geometry and the adapter rejects this in those modes.\n * `1080p` is grok-imagine-video-1.5 generation only; reference-to-video\n * is capped at 720p.\n */\n resolution?: GrokVideoResolution\n\n /**\n * Video duration in integer seconds (1–15). In `'extend'` mode this is\n * the length of the added tail only, not the total output length. Not\n * valid in `'edit'` mode (the output inherits the source clip's length).\n */\n duration?: number\n}\n\n/**\n * Provider options for grok-imagine-video (v1.0), which is the only model\n * that accepts a source-video edit / extend job.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GrokVideoSourceProviderOptions extends GrokVideoBaseProviderOptions {\n /**\n * Selects the request mode for a source-video prompt part: `'edit'`\n * (`/v1/videos/edits`) or `'extend'` (`/v1/videos/extensions`). Required\n * when the prompt carries a video part; not valid without one. Omit for\n * plain generation (`/v1/videos/generations`). grok-imagine-video only.\n */\n mode?: GrokVideoMode\n}\n\n/**\n * Provider options for grok-imagine-video-1.5, which adds the\n * reference-to-video inputs on top of the shared options.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GrokVideoProviderOptions extends GrokVideoBaseProviderOptions {\n /**\n * Reference images for reference-to-video generation (output capped at\n * 720p). Usually populated from image prompt parts with\n * `metadata.role: 'reference'` (or `'character'`); set explicitly to\n * replace the part-derived list. Reference images are addressed from the\n * prompt text as `<IMAGE_0>`, `<IMAGE_1>`, … in request order, and do not\n * lock the first frame.\n */\n reference_images?: Array<{ url: string }>\n\n /**\n * Preset TTS voices to reference for generated speech (max 3). Voice ids\n * come from the xAI TTS voice roster (e.g. 'eve', 'rex') or a custom\n * voice id, and are addressed from the prompt text as `<AUDIO_0>`,\n * `<AUDIO_1>`, `<AUDIO_2>`.\n */\n reference_audios?: Array<{ voice_id: string }>\n}\n\n/**\n * Widest option surface. Used when `modelOptions` arrives as deserialized\n * JSON and the adapter must validate fields the per-model map already\n * hides at compile time.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoRuntimeOptions = GrokVideoSourceProviderOptions &\n GrokVideoProviderOptions\n\n/**\n * Maximum reference voices accepted by the Imagine video endpoint.\n */\nexport const GROK_VIDEO_MAX_REFERENCE_AUDIOS = 3\n\n/**\n * Maximum reference images accepted by the Imagine video endpoint.\n */\nexport const GROK_VIDEO_MAX_REFERENCE_IMAGES = 7\n\n/**\n * Model names whose per-model options declare the reference fields. Keeps\n * the runtime set below provably in sync with\n * {@link GrokVideoModelProviderOptionsByName} — a typo or a new\n * reference-capable model missing from the set is a compile error.\n */\ntype GrokVideoReferenceModel = {\n [TModel in GrokVideoModel]: 'reference_images' extends keyof GrokVideoModelProviderOptionsByName[TModel]\n ? TModel\n : never\n}[GrokVideoModel]\n\n/**\n * Models that support reference-to-video inputs (`reference_images` /\n * `reference_audios`). The per-model provider-options map hides the fields\n * from other models at compile time; this backs the runtime gate for\n * untyped callers (e.g. deserialized JSON) so they get a clear error\n * instead of a raw API 400.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nconst GROK_VIDEO_REFERENCE_MODELS: ReadonlySet<string> =\n new Set<GrokVideoReferenceModel>(['grok-imagine-video-1.5'])\n\n/**\n * True when the model accepts reference-to-video inputs.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function isGrokVideoReferenceModel(model: string): boolean {\n return GROK_VIDEO_REFERENCE_MODELS.has(model)\n}\n\n/**\n * Model names whose per-model options declare `mode`. Same\n * provably-in-sync construction as {@link GrokVideoReferenceModel}.\n */\ntype GrokVideoSourceModel = {\n [TModel in GrokVideoModel]: 'mode' extends keyof GrokVideoModelProviderOptionsByName[TModel]\n ? TModel\n : never\n}[GrokVideoModel]\n\n/**\n * Models that accept a source-video prompt part for `/v1/videos/edits`\n * and `/v1/videos/extensions`. xAI lists video input only on\n * grok-imagine-video (v1.0).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nconst GROK_VIDEO_SOURCE_MODELS: ReadonlySet<string> =\n new Set<GrokVideoSourceModel>(['grok-imagine-video'])\n\n/**\n * True when the model accepts edit / extend source-video jobs.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function isGrokVideoSourceModel(model: string): boolean {\n return GROK_VIDEO_SOURCE_MODELS.has(model)\n}\n\n/**\n * Type-only map from model name to its specific provider options. Only\n * grok-imagine-video-1.5 exposes the reference-to-video fields. Only\n * grok-imagine-video (v1.0) exposes `mode` for edit / extend.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoModelProviderOptionsByName = {\n 'grok-imagine-video': GrokVideoSourceProviderOptions\n 'grok-imagine-video-1.5': GrokVideoProviderOptions\n}\n\n/**\n * Type-only map from model name to its supported `size` strings.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoModelSizeByName = {\n 'grok-imagine-video': GrokVideoSizeV1\n 'grok-imagine-video-1.5': GrokVideoSize\n}\n\n/**\n * Type-only map from model name to the non-text prompt modalities it accepts.\n * Both models support text-to-video and accept an optional `image` prompt\n * part as the starting frame; image parts with `metadata.role: 'reference'`\n * or `'character'` become `reference_images` (grok-imagine-video-1.5 only).\n * A `video` prompt part carries the source clip for edit / extension mode\n * on grok-imagine-video only (`modelOptions.mode: 'edit' | 'extend'`).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoModelInputModalitiesByName = {\n 'grok-imagine-video': readonly ['image', 'video']\n 'grok-imagine-video-1.5': readonly ['image']\n}\n"],"mappings":";AAoEA,IAAM,2BAAkD;CACtD;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,IAAM,yBAAgD;CAAC;CAAQ;CAAQ;AAAO;;;;;;;AAc9E,SAAgB,mBACd,MAC0D;CAE1D,MAAM,GAAG,aAAa,cADR,KAAK,MAAM,6BACW,KAAS,CAAC;CAC9C,IAAI,gBAAgB,KAAA,GAAW,OAAO,KAAA;CACtC,OAAO;EAAE;EAAa,GAAI,eAAe,KAAA,KAAa,EAAE,WAAW;CAAG;AACxE;;;;;;;AAQA,SAAgB,4BAA4B,OAAwB;CAClE,OAAO,UAAU;AACnB;;;;;;AAOA,SAAgB,kBACd,OACA,MAC2C;CAC3C,IAAI,SAAS,KAAA,GAAW;CACxB,MAAM,SAAS,mBAAmB,IAAI;CACtC,IAAI,CAAC,UAAU,CAAC,yBAAyB,SAAS,OAAO,WAAW,GAClE,MAAM,IAAI,MACR,SAAS,KAAK,+BAA+B,MAAM,qGAEzB,yBAAyB,KAAK,IAAI,GAC9D;CAEF,IACE,OAAO,eAAe,KAAA,KACtB,CAAC,uBAAuB,SAAS,OAAO,UAAU,GAElD,MAAM,IAAI,MACR,eAAe,OAAO,WAAW,+BAA+B,MAAM,4BAC1C,uBAAuB,KAAK,IAAI,GAC9D;CAEF,IAAI,OAAO,eAAe,WAAW,CAAC,4BAA4B,KAAK,GACrE,MAAM,IAAI,MACR,iDAAiD,MAAM,iFAEzD;AAEJ;;;;;;;;AAsBA,IAAa,uBAIT;CACF,sBAAsB;EACpB,MAAM;EACN,KAAA;EACA,KAAA;EACA,MAAM;EACN,MAAM;CACR;CACA,0BAA0B;EACxB,MAAM;EACN,KAAA;EACA,KAAA;EACA,MAAM;EACN,MAAM;CACR;AACF;;;;;;AAOA,SAAgB,4BACd,OACuD;CACvD,OAAO,qBAAqB;AAC9B;;;;;;;;;;AAoIA,IAAM,8CACJ,IAAI,IAA6B,CAAC,wBAAwB,CAAC;;;;;;AAO7D,SAAgB,0BAA0B,OAAwB;CAChE,OAAO,4BAA4B,IAAI,KAAK;AAC9C;;;;;;;;AAmBA,IAAM,2CACJ,IAAI,IAA0B,CAAC,oBAAoB,CAAC;;;;;;AAOtD,SAAgB,uBAAuB,OAAwB;CAC7D,OAAO,yBAAyB,IAAI,KAAK;AAC3C"}
1
+ {"version":3,"file":"video-provider-options.js","names":[],"sources":["../../../src/video/video-provider-options.ts"],"sourcesContent":["/**\n * Grok Video Generation Provider Options (xAI Imagine API)\n *\n * Based on https://docs.x.ai/developers/model-capabilities/video/generation\n * (plus the image-to-video, reference-to-video, editing, and extension pages\n * under the same section).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\n\nimport type { DurationOptions } from '@tanstack/ai/adapters'\nimport type { GrokVideoModel } from '../model-meta'\n\n/**\n * Aspect ratios accepted by the grok-imagine video models.\n *\n * Note: this is a narrower set than the grok-imagine image models — the\n * video endpoint rejects the phone-screen ratios ('9:19.5', '9:20', …) and\n * 'auto'.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoAspectRatio =\n | '1:1'\n | '16:9'\n | '9:16'\n | '4:3'\n | '3:4'\n | '3:2'\n | '2:3'\n\n/**\n * Resolution tiers for the grok-imagine video models.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoResolution = '480p' | '720p' | '1080p'\n\n/**\n * Resolutions accepted by grok-imagine-video (v1.0). Native 1080p is a\n * grok-imagine-video-1.5 generation feature.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoResolutionV1 = '480p' | '720p'\n\n/**\n * Size strings for grok-imagine video models. The Imagine API is\n * aspect-ratio based rather than pixel-size based; like the grok-imagine\n * image models, the generic `size` option uses an\n * `aspectRatio_resolution` template (\"16:9_720p\") — the resolution suffix\n * is optional (\"16:9\" uses the API default).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoSize =\n | GrokVideoAspectRatio\n | `${GrokVideoAspectRatio}_${GrokVideoResolution}`\n\n/**\n * Size strings for grok-imagine-video (v1.0) — 1080p is not in the type.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoSizeV1 =\n | GrokVideoAspectRatio\n | `${GrokVideoAspectRatio}_${GrokVideoResolutionV1}`\n\nconst GROK_VIDEO_ASPECT_RATIOS: ReadonlyArray<string> = [\n '1:1',\n '16:9',\n '9:16',\n '4:3',\n '3:4',\n '3:2',\n '2:3',\n]\n\nconst GROK_VIDEO_RESOLUTIONS: ReadonlyArray<string> = ['480p', '720p', '1080p']\n\n/**\n * Video duration limits enforced by the Imagine API (seconds).\n */\nexport const GROK_VIDEO_MIN_DURATION = 1\nexport const GROK_VIDEO_MAX_DURATION = 15\n\n/**\n * Parses a grok video size string into its components.\n * Format: \"aspectRatio\" or \"aspectRatio_resolution\",\n * e.g. \"16:9_720p\" → { aspectRatio: \"16:9\", resolution: \"720p\" }.\n * Returns undefined when the string doesn't match the template.\n */\nexport function parseGrokVideoSize(\n size: string,\n): { aspectRatio: string; resolution?: string } | undefined {\n const match = size.match(/^([\\d.]+:[\\d.]+)(?:_(.+))?$/)\n const [, aspectRatio, resolution] = match ?? []\n if (aspectRatio === undefined) return undefined\n return { aspectRatio, ...(resolution !== undefined && { resolution }) }\n}\n\n/**\n * Models that accept native 1080p on text-to-video and image-to-video.\n * Reference-to-video stays capped at 720p even on these models.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function isGrokVideoNative1080pModel(model: string): boolean {\n return model === 'grok-imagine-video-1.5'\n}\n\n/**\n * Validate the `size` template for a given grok video model.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function validateVideoSize(\n model: string,\n size?: string,\n): asserts size is GrokVideoSize | undefined {\n if (size === undefined) return\n const parsed = parseGrokVideoSize(size)\n if (!parsed || !GROK_VIDEO_ASPECT_RATIOS.includes(parsed.aspectRatio)) {\n throw new Error(\n `Size \"${size}\" is not supported by model \"${model}\". Expected ` +\n `\"aspectRatio\" or \"aspectRatio_resolution\" (e.g. \"16:9_720p\") with ` +\n `aspect ratio one of: ${GROK_VIDEO_ASPECT_RATIOS.join(', ')}`,\n )\n }\n if (\n parsed.resolution !== undefined &&\n !GROK_VIDEO_RESOLUTIONS.includes(parsed.resolution)\n ) {\n throw new Error(\n `Resolution \"${parsed.resolution}\" is not supported by model \"${model}\". ` +\n `Supported resolutions: ${GROK_VIDEO_RESOLUTIONS.join(', ')}`,\n )\n }\n if (parsed.resolution === '1080p' && !isGrokVideoNative1080pModel(model)) {\n throw new Error(\n `Resolution \"1080p\" is not supported by model \"${model}\". ` +\n `Use 'grok-imagine-video-1.5' for native 1080p text-to-video / image-to-video.`,\n )\n }\n}\n\n/**\n * Per-model duration type. The Imagine API accepts any integer second in the\n * 1–15 range, so this is a continuous range expressed as `number` (a literal\n * union can't represent it). `snapDuration()` coerces a raw seconds value into\n * the valid range at runtime.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoModelDurationByName = {\n 'grok-imagine-video': number\n 'grok-imagine-video-1.5': number\n}\n\n/**\n * Runtime duration table backing `availableDurations()` / `snapDuration()`.\n * Both grok-imagine video models accept the same continuous 1–15 integer-second\n * range.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport const GROK_VIDEO_DURATIONS: {\n readonly [TModel in GrokVideoModel]: DurationOptions<\n GrokVideoModelDurationByName[TModel]\n >\n} = {\n 'grok-imagine-video': {\n kind: 'range',\n min: GROK_VIDEO_MIN_DURATION,\n max: GROK_VIDEO_MAX_DURATION,\n step: 1,\n unit: 'seconds',\n },\n 'grok-imagine-video-1.5': {\n kind: 'range',\n min: GROK_VIDEO_MIN_DURATION,\n max: GROK_VIDEO_MAX_DURATION,\n step: 1,\n unit: 'seconds',\n },\n}\n\n/**\n * Look up the duration options for a grok video model.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function getGrokVideoDurationOptions<TModel extends GrokVideoModel>(\n model: TModel,\n): DurationOptions<GrokVideoModelDurationByName[TModel]> {\n return GROK_VIDEO_DURATIONS[model]\n}\n\n/**\n * Request mode for a source-video job. `'edit'` posts to `/v1/videos/edits`\n * (modify the source clip in place); `'extend'` posts to\n * `/v1/videos/extensions` (continue the source clip — `duration` is the\n * length of the **added tail**, not the total). Both require exactly one\n * video prompt part carrying the source clip, and both are\n * `grok-imagine-video` (v1.0) only.\n *\n * Output geometry (aspect ratio / resolution) is inherited from the source\n * clip in both modes, capped at 720p, and edit outputs also inherit the\n * source length — the adapter rejects `size`, `aspect_ratio`, `resolution`,\n * and (in edit mode) `duration` rather than sending fields the API ignores.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoMode = 'edit' | 'extend'\n\n/**\n * Provider options shared by both grok-imagine video models. These map\n * directly onto the Imagine API request body and take precedence over the\n * generic `size` / `duration` options when both are provided.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GrokVideoBaseProviderOptions {\n /**\n * Output aspect ratio. Generation only — edit / extend outputs inherit\n * the source clip's geometry and the adapter rejects this in those modes.\n */\n aspect_ratio?: GrokVideoAspectRatio\n\n /**\n * Output resolution tier. Generation only — edit / extend outputs inherit\n * the source clip's geometry and the adapter rejects this in those modes.\n * `1080p` is grok-imagine-video-1.5 generation only; reference-to-video\n * is capped at 720p.\n */\n resolution?: GrokVideoResolution\n\n /**\n * Video duration in integer seconds (1–15). In `'extend'` mode this is\n * the length of the added tail only, not the total output length. Not\n * valid in `'edit'` mode (the output inherits the source clip's length).\n */\n duration?: number\n}\n\n/**\n * Provider options for grok-imagine-video (v1.0), which is the only model\n * that accepts a source-video edit / extend job.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GrokVideoSourceProviderOptions extends GrokVideoBaseProviderOptions {\n /**\n * Selects the request mode for a source-video prompt part: `'edit'`\n * (`/v1/videos/edits`) or `'extend'` (`/v1/videos/extensions`). Required\n * when the prompt carries a video part; not valid without one. Omit for\n * plain generation (`/v1/videos/generations`). grok-imagine-video only.\n */\n mode?: GrokVideoMode\n}\n\n/**\n * Provider options for grok-imagine-video-1.5, which adds the\n * reference-to-video inputs on top of the shared options.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GrokVideoProviderOptions extends GrokVideoBaseProviderOptions {\n /**\n * Reference images for reference-to-video generation (output capped at\n * 720p). Usually populated from image prompt parts with\n * `metadata.role: 'reference'` (or `'character'`); set explicitly to\n * replace the part-derived list. Reference images are addressed from the\n * prompt text as `<IMAGE_0>`, `<IMAGE_1>`, … in request order, and do not\n * lock the first frame — pair them with a starting-frame image prompt part\n * when the opening frame has to be pinned.\n */\n reference_images?: Array<{ url: string }>\n\n /**\n * Preset TTS voices to reference for generated speech (max 3). Voice ids\n * come from the xAI TTS voice roster (e.g. 'eve', 'rex') or a custom\n * voice id, and are addressed from the prompt text as `<AUDIO_0>`,\n * `<AUDIO_1>`, `<AUDIO_2>`.\n */\n reference_audios?: Array<{ voice_id: string }>\n}\n\n/**\n * Widest option surface. Used when `modelOptions` arrives as deserialized\n * JSON and the adapter must validate fields the per-model map already\n * hides at compile time.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoRuntimeOptions = GrokVideoSourceProviderOptions &\n GrokVideoProviderOptions\n\n/**\n * Maximum reference voices accepted by the Imagine video endpoint.\n */\nexport const GROK_VIDEO_MAX_REFERENCE_AUDIOS = 3\n\n/**\n * Maximum reference images accepted by the Imagine video endpoint.\n */\nexport const GROK_VIDEO_MAX_REFERENCE_IMAGES = 7\n\n/**\n * Model names whose per-model options declare the reference fields. Keeps\n * the runtime set below provably in sync with\n * {@link GrokVideoModelProviderOptionsByName} — a typo or a new\n * reference-capable model missing from the set is a compile error.\n */\ntype GrokVideoReferenceModel = {\n [TModel in GrokVideoModel]: 'reference_images' extends keyof GrokVideoModelProviderOptionsByName[TModel]\n ? TModel\n : never\n}[GrokVideoModel]\n\n/**\n * Models that support reference-to-video inputs (`reference_images` /\n * `reference_audios`). The per-model provider-options map hides the fields\n * from other models at compile time; this backs the runtime gate for\n * untyped callers (e.g. deserialized JSON) so they get a clear error\n * instead of a raw API 400.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nconst GROK_VIDEO_REFERENCE_MODELS: ReadonlySet<string> =\n new Set<GrokVideoReferenceModel>(['grok-imagine-video-1.5'])\n\n/**\n * True when the model accepts reference-to-video inputs.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function isGrokVideoReferenceModel(model: string): boolean {\n return GROK_VIDEO_REFERENCE_MODELS.has(model)\n}\n\n/**\n * Model names whose per-model options declare `mode`. Same\n * provably-in-sync construction as {@link GrokVideoReferenceModel}.\n */\ntype GrokVideoSourceModel = {\n [TModel in GrokVideoModel]: 'mode' extends keyof GrokVideoModelProviderOptionsByName[TModel]\n ? TModel\n : never\n}[GrokVideoModel]\n\n/**\n * Models that accept a source-video prompt part for `/v1/videos/edits`\n * and `/v1/videos/extensions`. xAI lists video input only on\n * grok-imagine-video (v1.0).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nconst GROK_VIDEO_SOURCE_MODELS: ReadonlySet<string> =\n new Set<GrokVideoSourceModel>(['grok-imagine-video'])\n\n/**\n * True when the model accepts edit / extend source-video jobs.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport function isGrokVideoSourceModel(model: string): boolean {\n return GROK_VIDEO_SOURCE_MODELS.has(model)\n}\n\n/**\n * Type-only map from model name to its specific provider options. Only\n * grok-imagine-video-1.5 exposes the reference-to-video fields. Only\n * grok-imagine-video (v1.0) exposes `mode` for edit / extend.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoModelProviderOptionsByName = {\n 'grok-imagine-video': GrokVideoSourceProviderOptions\n 'grok-imagine-video-1.5': GrokVideoProviderOptions\n}\n\n/**\n * Type-only map from model name to its supported `size` strings.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoModelSizeByName = {\n 'grok-imagine-video': GrokVideoSizeV1\n 'grok-imagine-video-1.5': GrokVideoSize\n}\n\n/**\n * Type-only map from model name to the non-text prompt modalities it accepts.\n * Both models support text-to-video and accept an optional `image` prompt\n * part as the starting frame; image parts with `metadata.role: 'reference'`\n * or `'character'` become `reference_images` (grok-imagine-video-1.5 only).\n * On grok-imagine-video-1.5 the two combine: the starting frame pins the\n * first frame while the reference images steer subjects and style.\n * A `video` prompt part carries the source clip for edit / extension mode\n * on grok-imagine-video only (`modelOptions.mode: 'edit' | 'extend'`).\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport type GrokVideoModelInputModalitiesByName = {\n 'grok-imagine-video': readonly ['image', 'video']\n 'grok-imagine-video-1.5': readonly ['image']\n}\n"],"mappings":";AAoEA,IAAM,2BAAkD;CACtD;CACA;CACA;CACA;CACA;CACA;CACA;AACF;AAEA,IAAM,yBAAgD;CAAC;CAAQ;CAAQ;AAAO;;;;;;;AAc9E,SAAgB,mBACd,MAC0D;CAE1D,MAAM,GAAG,aAAa,cADR,KAAK,MAAM,6BACW,KAAS,CAAC;CAC9C,IAAI,gBAAgB,KAAA,GAAW,OAAO,KAAA;CACtC,OAAO;EAAE;EAAa,GAAI,eAAe,KAAA,KAAa,EAAE,WAAW;CAAG;AACxE;;;;;;;AAQA,SAAgB,4BAA4B,OAAwB;CAClE,OAAO,UAAU;AACnB;;;;;;AAOA,SAAgB,kBACd,OACA,MAC2C;CAC3C,IAAI,SAAS,KAAA,GAAW;CACxB,MAAM,SAAS,mBAAmB,IAAI;CACtC,IAAI,CAAC,UAAU,CAAC,yBAAyB,SAAS,OAAO,WAAW,GAClE,MAAM,IAAI,MACR,SAAS,KAAK,+BAA+B,MAAM,qGAEzB,yBAAyB,KAAK,IAAI,GAC9D;CAEF,IACE,OAAO,eAAe,KAAA,KACtB,CAAC,uBAAuB,SAAS,OAAO,UAAU,GAElD,MAAM,IAAI,MACR,eAAe,OAAO,WAAW,+BAA+B,MAAM,4BAC1C,uBAAuB,KAAK,IAAI,GAC9D;CAEF,IAAI,OAAO,eAAe,WAAW,CAAC,4BAA4B,KAAK,GACrE,MAAM,IAAI,MACR,iDAAiD,MAAM,iFAEzD;AAEJ;;;;;;;;AAsBA,IAAa,uBAIT;CACF,sBAAsB;EACpB,MAAM;EACN,KAAA;EACA,KAAA;EACA,MAAM;EACN,MAAM;CACR;CACA,0BAA0B;EACxB,MAAM;EACN,KAAA;EACA,KAAA;EACA,MAAM;EACN,MAAM;CACR;AACF;;;;;;AAOA,SAAgB,4BACd,OACuD;CACvD,OAAO,qBAAqB;AAC9B;;;;;;;;;;AAqIA,IAAM,8CACJ,IAAI,IAA6B,CAAC,wBAAwB,CAAC;;;;;;AAO7D,SAAgB,0BAA0B,OAAwB;CAChE,OAAO,4BAA4B,IAAI,KAAK;AAC9C;;;;;;;;AAmBA,IAAM,2CACJ,IAAI,IAA0B,CAAC,oBAAoB,CAAC;;;;;;AAOtD,SAAgB,uBAAuB,OAAwB;CAC7D,OAAO,yBAAyB,IAAI,KAAK;AAC3C"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai-grok",
3
- "version": "0.18.5",
3
+ "version": "0.18.7",
4
4
  "description": "xAI Grok adapter for TanStack AI chat, image generation, realtime, and structured outputs.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -58,18 +58,18 @@
58
58
  ],
59
59
  "dependencies": {
60
60
  "openai": "^6.41.0",
61
- "@tanstack/openai-base": "^0.10.11",
62
- "@tanstack/ai-utils": "^0.4.0"
61
+ "@tanstack/ai-utils": "^0.4.0",
62
+ "@tanstack/openai-base": "^0.10.12"
63
63
  },
64
64
  "devDependencies": {
65
65
  "@vitest/coverage-v8": "4.1.10",
66
66
  "google-auth-library": "^10.5.0",
67
67
  "vite": "^8.2.1",
68
- "@tanstack/ai": "0.54.0"
68
+ "@tanstack/ai": "0.55.0"
69
69
  },
70
70
  "peerDependencies": {
71
71
  "google-auth-library": "^10.5.0",
72
- "@tanstack/ai": "^0.54.0"
72
+ "@tanstack/ai": "^0.55.0"
73
73
  },
74
74
  "peerDependenciesMeta": {
75
75
  "google-auth-library": {
@@ -363,17 +363,13 @@ export class GrokVideoAdapter<
363
363
  // Image-to-video: the single image prompt part becomes the starting frame
364
364
  // and the prompt text describes the desired motion. URL sources are
365
365
  // fetched by xAI's servers; data sources are sent as base64 data URIs.
366
+ // On grok-imagine-video-1.5 the starting frame combines with reference
367
+ // inputs — `image` pins the first frame while reference_images /
368
+ // reference_audios steer subjects and voices. Classic grok-imagine-video
369
+ // rejects that mix, but it rejects reference inputs outright, so the
370
+ // model gate above already covers it.
366
371
  const [startFrame] = startFrames
367
372
 
368
- // xAI rejects `image` + `reference_images` / `reference_audios` as a
369
- // 400: only one of image-to-video or reference-to-video can be active.
370
- if (startFrame && hasReference) {
371
- throw new Error(
372
- `${this.name}: image-to-video and reference-to-video cannot be combined. ` +
373
- `Use a starting-frame image, or reference images / voices, not both.`,
374
- )
375
- }
376
-
377
373
  // The generic `size` option carries an "aspectRatio_resolution" template
378
374
  // (e.g. '16:9_720p') and maps to the Imagine API's `aspect_ratio` /
379
375
  // `resolution` parameters; explicit modelOptions win over the template
@@ -272,7 +272,8 @@ export interface GrokVideoProviderOptions extends GrokVideoBaseProviderOptions {
272
272
  * `metadata.role: 'reference'` (or `'character'`); set explicitly to
273
273
  * replace the part-derived list. Reference images are addressed from the
274
274
  * prompt text as `<IMAGE_0>`, `<IMAGE_1>`, … in request order, and do not
275
- * lock the first frame.
275
+ * lock the first frame — pair them with a starting-frame image prompt part
276
+ * when the opening frame has to be pinned.
276
277
  */
277
278
  reference_images?: Array<{ url: string }>
278
279
 
@@ -394,6 +395,8 @@ export type GrokVideoModelSizeByName = {
394
395
  * Both models support text-to-video and accept an optional `image` prompt
395
396
  * part as the starting frame; image parts with `metadata.role: 'reference'`
396
397
  * or `'character'` become `reference_images` (grok-imagine-video-1.5 only).
398
+ * On grok-imagine-video-1.5 the two combine: the starting frame pins the
399
+ * first frame while the reference images steer subjects and style.
397
400
  * A `video` prompt part carries the source clip for edit / extension mode
398
401
  * on grok-imagine-video only (`modelOptions.mode: 'edit' | 'extend'`).
399
402
  *