@tanstack/ai-gemini 0.19.1 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/dist/esm/adapters/audio.d.ts +1 -1
  2. package/dist/esm/adapters/audio.js.map +1 -1
  3. package/dist/esm/adapters/image.d.ts +1 -1
  4. package/dist/esm/adapters/image.js +17 -39
  5. package/dist/esm/adapters/image.js.map +1 -1
  6. package/dist/esm/adapters/summarize.d.ts +1 -1
  7. package/dist/esm/adapters/summarize.js.map +1 -1
  8. package/dist/esm/adapters/text.d.ts +1 -1
  9. package/dist/esm/adapters/text.js.map +1 -1
  10. package/dist/esm/adapters/tts.d.ts +1 -1
  11. package/dist/esm/adapters/tts.js.map +1 -1
  12. package/dist/esm/adapters/video.d.ts +60 -11
  13. package/dist/esm/adapters/video.js +205 -6
  14. package/dist/esm/adapters/video.js.map +1 -1
  15. package/dist/esm/experimental/text-interactions/adapter.d.ts +1 -1
  16. package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
  17. package/dist/esm/index.d.ts +6 -3
  18. package/dist/esm/index.js +9 -3
  19. package/dist/esm/index.js.map +1 -1
  20. package/dist/esm/model-meta.d.ts +11 -3
  21. package/dist/esm/model-meta.js +9 -1
  22. package/dist/esm/model-meta.js.map +1 -1
  23. package/dist/esm/realtime/adapter.d.ts +22 -0
  24. package/dist/esm/realtime/adapter.js +233 -0
  25. package/dist/esm/realtime/adapter.js.map +1 -0
  26. package/dist/esm/realtime/client.d.ts +98 -0
  27. package/dist/esm/realtime/client.js +389 -0
  28. package/dist/esm/realtime/client.js.map +1 -0
  29. package/dist/esm/realtime/index.d.ts +3 -0
  30. package/dist/esm/realtime/token.d.ts +26 -0
  31. package/dist/esm/realtime/token.js +39 -0
  32. package/dist/esm/realtime/token.js.map +1 -0
  33. package/dist/esm/realtime/types.d.ts +51 -0
  34. package/dist/esm/realtime/utils.d.ts +40 -0
  35. package/dist/esm/realtime/utils.js +350 -0
  36. package/dist/esm/realtime/utils.js.map +1 -0
  37. package/dist/esm/video/video-provider-options.d.ts +59 -14
  38. package/dist/esm/video/video-provider-options.js +15 -2
  39. package/dist/esm/video/video-provider-options.js.map +1 -1
  40. package/package.json +4 -4
  41. package/src/adapters/audio.ts +1 -1
  42. package/src/adapters/image.ts +25 -49
  43. package/src/adapters/summarize.ts +1 -1
  44. package/src/adapters/text.ts +1 -1
  45. package/src/adapters/tts.ts +1 -1
  46. package/src/adapters/video.ts +333 -16
  47. package/src/experimental/text-interactions/adapter.ts +2 -2
  48. package/src/index.ts +20 -2
  49. package/src/model-meta.ts +45 -2
  50. package/src/realtime/adapter.ts +311 -0
  51. package/src/realtime/client.ts +547 -0
  52. package/src/realtime/index.ts +14 -0
  53. package/src/realtime/token.ts +70 -0
  54. package/src/realtime/types.ts +94 -0
  55. package/src/realtime/utils.ts +439 -0
  56. package/src/video/video-provider-options.ts +95 -15
@@ -1 +1 @@
1
- {"version":3,"file":"tts.js","sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport {\n createGeminiClient,\n generateId,\n getGeminiApiKeyFromEnv,\n} from '../utils'\nimport { GEMINI_TTS_VOICES } from '../model-meta'\nimport { buildGeminiUsage } from '../usage'\nimport type { GEMINI_TTS_MODELS, GeminiTTSVoice } from '../model-meta'\nimport type { TTSOptions, TTSResult } from '@tanstack/ai'\nimport type { GoogleGenAI, SpeechConfig } from '@google/genai'\nimport type { GeminiClientConfig } from '../utils'\n\n/**\n * Configuration for a single speaker in a multi-speaker dialogue.\n * Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.\n */\nexport interface GeminiSpeakerVoiceConfig {\n /** A name used in the prompt to refer to this speaker */\n speaker: string\n /** Voice configuration for this speaker */\n voiceConfig: {\n prebuiltVoiceConfig: {\n voiceName: GeminiTTSVoice\n }\n }\n}\n\n/**\n * Provider-specific options for Gemini TTS\n *\n * @experimental Gemini TTS is an experimental feature.\n * @see https://ai.google.dev/gemini-api/docs/speech-generation\n * @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview\n */\nexport interface GeminiTTSProviderOptions {\n /**\n * Voice configuration for single-speaker TTS.\n * Choose from 30 available voices with different characteristics.\n *\n * Use `multiSpeakerVoiceConfig` instead for dialogues.\n */\n voiceConfig?: {\n prebuiltVoiceConfig?: {\n /**\n * The voice name to use for speech synthesis.\n * @see https://ai.google.dev/gemini-api/docs/speech-generation#voices\n */\n voiceName?: GeminiTTSVoice\n }\n }\n\n /**\n * Multi-speaker voice configuration (up to 2 speakers).\n * Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.\n *\n * Each speaker's lines in the prompt are prefixed with the name defined\n * here, e.g.:\n *\n * ```text\n * Joe: Hey, how's it going?\n * Jane: Not bad, you?\n * ```\n */\n multiSpeakerVoiceConfig?: {\n speakerVoiceConfigs: Array<GeminiSpeakerVoiceConfig>\n }\n\n /**\n * System instruction for controlling speech style.\n * Use natural language to describe the desired speaking style,\n * pace, tone, accent, or other characteristics.\n *\n * With Gemini 3.1 Flash TTS, you can also use inline audio tags like\n * `[whispering]`, `[laughs]`, `[excited]` directly in the input text\n * to control delivery.\n *\n * @example \"Speak slowly and calmly, as if telling a bedtime story\"\n * @example \"Use an upbeat, enthusiastic tone with moderate pace\"\n * @example \"Speak with a British accent\"\n */\n systemInstruction?: string\n\n /**\n * Language code hint for the speech synthesis.\n * Gemini 3.1 Flash TTS supports 70+ languages with auto-detection;\n * the 2.5 TTS models support 24 languages.\n *\n * @example \"en-US\" for American English\n * @example \"es-ES\" for Spanish (Spain)\n * @example \"ja-JP\" for Japanese\n */\n languageCode?: string\n}\n\n/**\n * Configuration for Gemini TTS adapter\n *\n * @experimental Gemini TTS is an experimental feature.\n */\nexport interface GeminiTTSConfig extends GeminiClientConfig {}\n\n/** Model type for Gemini TTS */\nexport type GeminiTTSModel = (typeof GEMINI_TTS_MODELS)[number]\n\n/**\n * Gemini Text-to-Speech Adapter\n *\n * Tree-shakeable adapter for Gemini TTS functionality.\n *\n * **IMPORTANT**: Gemini TTS uses the Live API (WebSocket-based) which requires\n * different handling than traditional REST APIs. This adapter provides a\n * simplified interface but may have limitations.\n *\n * @experimental Gemini TTS is an experimental feature and may change.\n *\n * Models:\n * - gemini-2.5-flash-preview-tts\n */\nexport class GeminiTTSAdapter<\n TModel extends GeminiTTSModel,\n> extends BaseTTSAdapter<TModel, GeminiTTSProviderOptions> {\n readonly name = 'gemini' as const\n\n private readonly client: GoogleGenAI\n\n constructor(config: GeminiTTSConfig, model: TModel) {\n super(model, config)\n this.client = createGeminiClient(config)\n }\n\n /**\n * Generate speech from text using Gemini's TTS model.\n *\n * @experimental This implementation is experimental and may change.\n * @see https://ai.google.dev/gemini-api/docs/speech-generation\n */\n async generateSpeech(\n options: TTSOptions<GeminiTTSProviderOptions>,\n ): Promise<TTSResult> {\n const { model, text, modelOptions, voice, logger } = options\n\n logger.request(`activity=generateSpeech provider=gemini model=${model}`, {\n provider: 'gemini',\n model,\n })\n\n const speechConfig: SpeechConfig = {}\n\n if (modelOptions?.multiSpeakerVoiceConfig) {\n // Validate multi-speaker config: 1 or 2 speakers allowed.\n const speakerConfigs =\n modelOptions.multiSpeakerVoiceConfig.speakerVoiceConfigs\n if (\n !Array.isArray(speakerConfigs) ||\n speakerConfigs.length < 1 ||\n speakerConfigs.length > 2\n ) {\n throw new Error(\n `Gemini TTS multiSpeakerVoiceConfig.speakerVoiceConfigs must contain 1 or 2 speakers; received ${Array.isArray(speakerConfigs) ? speakerConfigs.length : 'non-array'}.`,\n )\n }\n speechConfig.multiSpeakerVoiceConfig =\n modelOptions.multiSpeakerVoiceConfig\n } else {\n // Honor the standard TTSOptions.voice (used by every other TTS adapter)\n // as a fallback for the prebuilt voice name. If an explicit\n // modelOptions.voiceConfig is supplied its values win — but we still\n // fall back to `voice` / 'Kore' if the supplied voiceConfig is missing\n // prebuiltVoiceConfig.voiceName.\n if (\n voice !== undefined &&\n !(GEMINI_TTS_VOICES as ReadonlyArray<string>).includes(voice)\n ) {\n throw new Error(\n `Invalid Gemini TTS voice \"${voice}\". Valid voices are: ${GEMINI_TTS_VOICES.join(', ')}.`,\n )\n }\n const defaultVoiceName = (voice as GeminiTTSVoice | undefined) ?? 'Kore'\n const supplied = modelOptions?.voiceConfig\n const resolvedVoiceName =\n supplied?.prebuiltVoiceConfig?.voiceName ?? defaultVoiceName\n speechConfig.voiceConfig = {\n prebuiltVoiceConfig: { voiceName: resolvedVoiceName },\n }\n }\n\n if (modelOptions?.languageCode) {\n speechConfig.languageCode = modelOptions.languageCode\n }\n\n try {\n const response = await this.client.models.generateContent({\n model,\n contents: [\n {\n role: 'user',\n parts: [{ text }],\n },\n ],\n config: {\n responseModalities: ['AUDIO'],\n speechConfig,\n // systemInstruction belongs inside `config` per the @google/genai\n // contract — matches sibling Gemini adapters (summarize, text).\n ...(modelOptions?.systemInstruction && {\n systemInstruction: modelOptions.systemInstruction,\n }),\n },\n })\n\n // Extract audio data from response\n const candidate = response.candidates?.[0]\n const parts = candidate?.content?.parts\n\n if (!parts || parts.length === 0) {\n throw new Error('No audio output received from Gemini TTS')\n }\n\n // Look for inline data (audio)\n const audioPart = parts.find((part: any) =>\n part.inlineData?.mimeType?.startsWith('audio/'),\n )\n\n if (!audioPart || !audioPart.inlineData || !audioPart.inlineData.data) {\n throw new Error('No audio data in Gemini TTS response')\n }\n\n const audioBase64 = audioPart.inlineData.data\n // mime is guaranteed by the `startsWith('audio/')` find predicate above.\n const mimeType = audioPart.inlineData.mimeType as string\n\n // Surface token usage (with per-modality breakdown) when Gemini reports\n // it. Spread conditionally for exactOptionalPropertyTypes — shared by both\n // the PCM→WAV and pass-through return paths below.\n const usageField = response.usageMetadata\n ? { usage: buildGeminiUsage(response.usageMetadata) }\n : {}\n\n // Gemini TTS models return raw 16-bit LE PCM with a mime type like\n // `audio/L16;codec=pcm;rate=24000`. That isn't playable in an <audio>\n // element and the bare string isn't a usable file extension, so we\n // prepend a RIFF/WAV header here and normalize the result to audio/wav.\n const pcm = parsePcmMimeType(mimeType)\n if (pcm) {\n const wavBase64 = wrapPcmBase64AsWav(\n audioBase64,\n pcm.sampleRate,\n pcm.channels,\n pcm.bitsPerSample,\n )\n return {\n id: generateId(this.name),\n model,\n audio: wavBase64,\n format: 'wav',\n contentType: 'audio/wav',\n ...usageField,\n }\n }\n\n // Strip any mime parameters (e.g. `audio/ogg;codec=opus`) before pulling\n // the subtype out as the file format. `String.split` always returns at\n // least one element, so [0] is defined.\n const format = (mimeType.split(';')[0] ?? '').split('/')[1] || 'wav'\n\n return {\n id: generateId(this.name),\n model,\n audio: audioBase64,\n format,\n contentType: mimeType,\n ...usageField,\n }\n } catch (error) {\n logger.errors('gemini.generateSpeech fatal', {\n error,\n source: 'gemini.generateSpeech',\n })\n throw error\n }\n }\n}\n\nfunction parsePcmMimeType(\n mimeType: string,\n): { sampleRate: number; channels: number; bitsPerSample: number } | undefined {\n const normalized = mimeType.toLowerCase()\n const subtype = (normalized.split(';')[0] ?? '').split('/')[1] ?? ''\n // Exclude containerized wav (e.g. `audio/wav;codec=pcm`) — those already\n // carry a RIFF header and must not be re-wrapped.\n if (subtype.includes('wav')) return undefined\n\n // Accept the variants Gemini and other providers actually emit:\n // - audio/L16;codec=pcm;rate=24000 (IANA PCM with bit depth in the type)\n // - audio/L24 and friends\n // - audio/pcm and audio/x-pcm\n // - anything else that explicitly tags codec=pcm and isn't wav-containered\n const bitDepthMatch = /^audio\\/l(\\d+)/.exec(normalized)\n const isPcm =\n bitDepthMatch !== null ||\n normalized.startsWith('audio/pcm') ||\n normalized.startsWith('audio/x-pcm') ||\n normalized.includes('codec=pcm')\n if (!isPcm) return undefined\n\n const rateMatch = /rate=(\\d+)/.exec(normalized)\n const channelsMatch = /channels=(\\d+)/.exec(normalized)\n // Default to 16-bit when the mime type doesn't specify — matches Gemini's\n // audio/L16;codec=pcm;rate=24000 response.\n const bitsPerSample = bitDepthMatch ? Number(bitDepthMatch[1]) : 16\n return {\n sampleRate: rateMatch ? Number(rateMatch[1]) : 24000,\n channels: channelsMatch ? Number(channelsMatch[1]) : 1,\n bitsPerSample,\n }\n}\n\nfunction wrapPcmBase64AsWav(\n pcmBase64: string,\n sampleRate: number,\n channels = 1,\n bitsPerSample = 16,\n): string {\n // The WAV writer below emits a 16-bit PCM fmt chunk. If the source claims a\n // different bit depth we'd be lying about the payload, so bail out loudly\n // rather than producing a corrupt file.\n if (bitsPerSample !== 16) {\n throw new Error(\n `Unsupported PCM bit depth ${bitsPerSample}: only 16-bit PCM can be wrapped as WAV.`,\n )\n }\n\n const pcmBytes =\n typeof Buffer !== 'undefined'\n ? new Uint8Array(Buffer.from(pcmBase64, 'base64'))\n : decodeBase64(pcmBase64)\n\n const byteRate = (sampleRate * channels * bitsPerSample) / 8\n const blockAlign = (channels * bitsPerSample) / 8\n const dataSize = pcmBytes.byteLength\n const buffer = new ArrayBuffer(44 + dataSize)\n const view = new DataView(buffer)\n\n writeAscii(view, 0, 'RIFF')\n view.setUint32(4, 36 + dataSize, true)\n writeAscii(view, 8, 'WAVE')\n writeAscii(view, 12, 'fmt ')\n view.setUint32(16, 16, true)\n view.setUint16(20, 1, true)\n view.setUint16(22, channels, true)\n view.setUint32(24, sampleRate, true)\n view.setUint32(28, byteRate, true)\n view.setUint16(32, blockAlign, true)\n view.setUint16(34, bitsPerSample, true)\n writeAscii(view, 36, 'data')\n view.setUint32(40, dataSize, true)\n new Uint8Array(buffer, 44).set(pcmBytes)\n\n if (typeof Buffer !== 'undefined') {\n return Buffer.from(buffer).toString('base64')\n }\n let binary = ''\n const bytes = new Uint8Array(buffer)\n for (const byte of bytes) {\n binary += String.fromCharCode(byte)\n }\n return btoa(binary)\n}\n\nfunction decodeBase64(b64: string): Uint8Array {\n const binary = atob(b64)\n const out = new Uint8Array(binary.length)\n for (let i = 0; i < binary.length; i += 1) out[i] = binary.charCodeAt(i)\n return out\n}\n\nfunction writeAscii(view: DataView, offset: number, text: string): void {\n for (let i = 0; i < text.length; i += 1) {\n view.setUint8(offset + i, text.charCodeAt(i))\n }\n}\n\n/**\n * Creates a Gemini TTS adapter with explicit API key.\n * Type resolution happens here at the call site.\n *\n * @experimental Gemini TTS is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'gemini-2.5-flash-preview-tts')\n * @param apiKey - Your Google API key\n * @param config - Optional additional configuration\n * @returns Configured Gemini TTS adapter instance with resolved types\n *\n * @example\n * ```typescript\n * const adapter = createGeminiSpeech('gemini-2.5-flash-preview-tts', \"your-api-key\");\n *\n * const result = await generateSpeech({\n * adapter,\n * text: 'Hello, world!'\n * });\n * ```\n */\nexport function createGeminiSpeech<TModel extends GeminiTTSModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GeminiTTSConfig, 'apiKey'>,\n): GeminiTTSAdapter<TModel> {\n // Put apiKey LAST so caller-supplied config can't silently override the\n // explicit argument.\n return new GeminiTTSAdapter({ ...config, apiKey }, model)\n}\n\n/**\n * Creates a Gemini speech adapter with automatic API key detection from environment variables.\n * Type resolution happens here at the call site.\n *\n * @experimental Gemini TTS is an experimental feature and may change.\n *\n * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (Browser with injected env)\n *\n * @param model - The model name (e.g., 'gemini-2.5-flash-preview-tts')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Gemini speech adapter instance with resolved types\n * @throws Error if GOOGLE_API_KEY or GEMINI_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * // Automatically uses GOOGLE_API_KEY from environment\n * const adapter = geminiSpeech('gemini-2.5-flash-preview-tts');\n *\n * const result = await generateSpeech({\n * adapter,\n * text: 'Welcome to TanStack AI!'\n * });\n * ```\n */\nexport function geminiSpeech<TModel extends GeminiTTSModel>(\n model: TModel,\n config?: Omit<GeminiTTSConfig, 'apiKey'>,\n): GeminiTTSAdapter<TModel> {\n const apiKey = getGeminiApiKeyFromEnv()\n return createGeminiSpeech(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;;AAuHO,MAAM,yBAEH,eAAiD;AAAA,EAChD,OAAO;AAAA,EAEC;AAAA,EAEjB,YAAY,QAAyB,OAAe;AAClD,UAAM,OAAO,MAAM;AACnB,SAAK,SAAS,mBAAmB,MAAM;AAAA,EACzC;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAQA,MAAM,eACJ,SACoB;AACpB,UAAM,EAAE,OAAO,MAAM,cAAc,OAAO,WAAW;AAErD,WAAO,QAAQ,iDAAiD,KAAK,IAAI;AAAA,MACvE,UAAU;AAAA,MACV;AAAA,IAAA,CACD;AAED,UAAM,eAA6B,CAAA;AAEnC,QAAI,cAAc,yBAAyB;AAEzC,YAAM,iBACJ,aAAa,wBAAwB;AACvC,UACE,CAAC,MAAM,QAAQ,cAAc,KAC7B,eAAe,SAAS,KACxB,eAAe,SAAS,GACxB;AACA,cAAM,IAAI;AAAA,UACR,iGAAiG,MAAM,QAAQ,cAAc,IAAI,eAAe,SAAS,WAAW;AAAA,QAAA;AAAA,MAExK;AACA,mBAAa,0BACX,aAAa;AAAA,IACjB,OAAO;AAML,UACE,UAAU,UACV,CAAE,kBAA4C,SAAS,KAAK,GAC5D;AACA,cAAM,IAAI;AAAA,UACR,6BAA6B,KAAK,wBAAwB,kBAAkB,KAAK,IAAI,CAAC;AAAA,QAAA;AAAA,MAE1F;AACA,YAAM,mBAAoB,SAAwC;AAClE,YAAM,WAAW,cAAc;AAC/B,YAAM,oBACJ,UAAU,qBAAqB,aAAa;AAC9C,mBAAa,cAAc;AAAA,QACzB,qBAAqB,EAAE,WAAW,kBAAA;AAAA,MAAkB;AAAA,IAExD;AAEA,QAAI,cAAc,cAAc;AAC9B,mBAAa,eAAe,aAAa;AAAA,IAC3C;AAEA,QAAI;AACF,YAAM,WAAW,MAAM,KAAK,OAAO,OAAO,gBAAgB;AAAA,QACxD;AAAA,QACA,UAAU;AAAA,UACR;AAAA,YACE,MAAM;AAAA,YACN,OAAO,CAAC,EAAE,KAAA,CAAM;AAAA,UAAA;AAAA,QAClB;AAAA,QAEF,QAAQ;AAAA,UACN,oBAAoB,CAAC,OAAO;AAAA,UAC5B;AAAA;AAAA;AAAA,UAGA,GAAI,cAAc,qBAAqB;AAAA,YACrC,mBAAmB,aAAa;AAAA,UAAA;AAAA,QAClC;AAAA,MACF,CACD;AAGD,YAAM,YAAY,SAAS,aAAa,CAAC;AACzC,YAAM,QAAQ,WAAW,SAAS;AAElC,UAAI,CAAC,SAAS,MAAM,WAAW,GAAG;AAChC,cAAM,IAAI,MAAM,0CAA0C;AAAA,MAC5D;AAGA,YAAM,YAAY,MAAM;AAAA,QAAK,CAAC,SAC5B,KAAK,YAAY,UAAU,WAAW,QAAQ;AAAA,MAAA;AAGhD,UAAI,CAAC,aAAa,CAAC,UAAU,cAAc,CAAC,UAAU,WAAW,MAAM;AACrE,cAAM,IAAI,MAAM,sCAAsC;AAAA,MACxD;AAEA,YAAM,cAAc,UAAU,WAAW;AAEzC,YAAM,WAAW,UAAU,WAAW;AAKtC,YAAM,aAAa,SAAS,gBACxB,EAAE,OAAO,iBAAiB,SAAS,aAAa,EAAA,IAChD,CAAA;AAMJ,YAAM,MAAM,iBAAiB,QAAQ;AACrC,UAAI,KAAK;AACP,cAAM,YAAY;AAAA,UAChB;AAAA,UACA,IAAI;AAAA,UACJ,IAAI;AAAA,UACJ,IAAI;AAAA,QAAA;AAEN,eAAO;AAAA,UACL,IAAI,WAAW,KAAK,IAAI;AAAA,UACxB;AAAA,UACA,OAAO;AAAA,UACP,QAAQ;AAAA,UACR,aAAa;AAAA,UACb,GAAG;AAAA,QAAA;AAAA,MAEP;AAKA,YAAM,UAAU,SAAS,MAAM,GAAG,EAAE,CAAC,KAAK,IAAI,MAAM,GAAG,EAAE,CAAC,KAAK;AAE/D,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA,OAAO;AAAA,QACP;AAAA,QACA,aAAa;AAAA,QACb,GAAG;AAAA,MAAA;AAAA,IAEP,SAAS,OAAO;AACd,aAAO,OAAO,+BAA+B;AAAA,QAC3C;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AAEA,SAAS,iBACP,UAC6E;AAC7E,QAAM,aAAa,SAAS,YAAA;AAC5B,QAAM,WAAW,WAAW,MAAM,GAAG,EAAE,CAAC,KAAK,IAAI,MAAM,GAAG,EAAE,CAAC,KAAK;AAGlE,MAAI,QAAQ,SAAS,KAAK,EAAG,QAAO;AAOpC,QAAM,gBAAgB,iBAAiB,KAAK,UAAU;AACtD,QAAM,QACJ,kBAAkB,QAClB,WAAW,WAAW,WAAW,KACjC,WAAW,WAAW,aAAa,KACnC,WAAW,SAAS,WAAW;AACjC,MAAI,CAAC,MAAO,QAAO;AAEnB,QAAM,YAAY,aAAa,KAAK,UAAU;AAC9C,QAAM,gBAAgB,iBAAiB,KAAK,UAAU;AAGtD,QAAM,gBAAgB,gBAAgB,OAAO,cAAc,CAAC,CAAC,IAAI;AACjE,SAAO;AAAA,IACL,YAAY,YAAY,OAAO,UAAU,CAAC,CAAC,IAAI;AAAA,IAC/C,UAAU,gBAAgB,OAAO,cAAc,CAAC,CAAC,IAAI;AAAA,IACrD;AAAA,EAAA;AAEJ;AAEA,SAAS,mBACP,WACA,YACA,WAAW,GACX,gBAAgB,IACR;AAIR,MAAI,kBAAkB,IAAI;AACxB,UAAM,IAAI;AAAA,MACR,6BAA6B,aAAa;AAAA,IAAA;AAAA,EAE9C;AAEA,QAAM,WACJ,OAAO,WAAW,cACd,IAAI,WAAW,OAAO,KAAK,WAAW,QAAQ,CAAC,IAC/C,aAAa,SAAS;AAE5B,QAAM,WAAY,aAAa,WAAW,gBAAiB;AAC3D,QAAM,aAAc,WAAW,gBAAiB;AAChD,QAAM,WAAW,SAAS;AAC1B,QAAM,SAAS,IAAI,YAAY,KAAK,QAAQ;AAC5C,QAAM,OAAO,IAAI,SAAS,MAAM;AAEhC,aAAW,MAAM,GAAG,MAAM;AAC1B,OAAK,UAAU,GAAG,KAAK,UAAU,IAAI;AACrC,aAAW,MAAM,GAAG,MAAM;AAC1B,aAAW,MAAM,IAAI,MAAM;AAC3B,OAAK,UAAU,IAAI,IAAI,IAAI;AAC3B,OAAK,UAAU,IAAI,GAAG,IAAI;AAC1B,OAAK,UAAU,IAAI,UAAU,IAAI;AACjC,OAAK,UAAU,IAAI,YAAY,IAAI;AACnC,OAAK,UAAU,IAAI,UAAU,IAAI;AACjC,OAAK,UAAU,IAAI,YAAY,IAAI;AACnC,OAAK,UAAU,IAAI,eAAe,IAAI;AACtC,aAAW,MAAM,IAAI,MAAM;AAC3B,OAAK,UAAU,IAAI,UAAU,IAAI;AACjC,MAAI,WAAW,QAAQ,EAAE,EAAE,IAAI,QAAQ;AAEvC,MAAI,OAAO,WAAW,aAAa;AACjC,WAAO,OAAO,KAAK,MAAM,EAAE,SAAS,QAAQ;AAAA,EAC9C;AACA,MAAI,SAAS;AACb,QAAM,QAAQ,IAAI,WAAW,MAAM;AACnC,aAAW,QAAQ,OAAO;AACxB,cAAU,OAAO,aAAa,IAAI;AAAA,EACpC;AACA,SAAO,KAAK,MAAM;AACpB;AAEA,SAAS,aAAa,KAAyB;AAC7C,QAAM,SAAS,KAAK,GAAG;AACvB,QAAM,MAAM,IAAI,WAAW,OAAO,MAAM;AACxC,WAAS,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK,EAAG,KAAI,CAAC,IAAI,OAAO,WAAW,CAAC;AACvE,SAAO;AACT;AAEA,SAAS,WAAW,MAAgB,QAAgB,MAAoB;AACtE,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK,GAAG;AACvC,SAAK,SAAS,SAAS,GAAG,KAAK,WAAW,CAAC,CAAC;AAAA,EAC9C;AACF;AAuBO,SAAS,mBACd,OACA,QACA,QAC0B;AAG1B,SAAO,IAAI,iBAAiB,EAAE,GAAG,QAAQ,OAAA,GAAU,KAAK;AAC1D;AA4BO,SAAS,aACd,OACA,QAC0B;AAC1B,QAAM,SAAS,uBAAA;AACf,SAAO,mBAAmB,OAAO,QAAQ,MAAM;AACjD;"}
1
+ {"version":3,"file":"tts.js","sources":["../../../src/adapters/tts.ts"],"sourcesContent":["import { BaseTTSAdapter } from '@tanstack/ai/adapters'\nimport {\n createGeminiClient,\n generateId,\n getGeminiApiKeyFromEnv,\n} from '../utils'\nimport { GEMINI_TTS_VOICES } from '../model-meta'\nimport { buildGeminiUsage } from '../usage'\nimport type { GEMINI_TTS_MODELS, GeminiTTSVoice } from '../model-meta'\nimport type { TTSOptions, TTSResult } from '@tanstack/ai'\nimport type { GoogleGenAI, SpeechConfig } from '@google/genai'\nimport type { GeminiClientConfig } from '../utils/client'\n\n/**\n * Configuration for a single speaker in a multi-speaker dialogue.\n * Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.\n */\nexport interface GeminiSpeakerVoiceConfig {\n /** A name used in the prompt to refer to this speaker */\n speaker: string\n /** Voice configuration for this speaker */\n voiceConfig: {\n prebuiltVoiceConfig: {\n voiceName: GeminiTTSVoice\n }\n }\n}\n\n/**\n * Provider-specific options for Gemini TTS\n *\n * @experimental Gemini TTS is an experimental feature.\n * @see https://ai.google.dev/gemini-api/docs/speech-generation\n * @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview\n */\nexport interface GeminiTTSProviderOptions {\n /**\n * Voice configuration for single-speaker TTS.\n * Choose from 30 available voices with different characteristics.\n *\n * Use `multiSpeakerVoiceConfig` instead for dialogues.\n */\n voiceConfig?: {\n prebuiltVoiceConfig?: {\n /**\n * The voice name to use for speech synthesis.\n * @see https://ai.google.dev/gemini-api/docs/speech-generation#voices\n */\n voiceName?: GeminiTTSVoice\n }\n }\n\n /**\n * Multi-speaker voice configuration (up to 2 speakers).\n * Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.\n *\n * Each speaker's lines in the prompt are prefixed with the name defined\n * here, e.g.:\n *\n * ```text\n * Joe: Hey, how's it going?\n * Jane: Not bad, you?\n * ```\n */\n multiSpeakerVoiceConfig?: {\n speakerVoiceConfigs: Array<GeminiSpeakerVoiceConfig>\n }\n\n /**\n * System instruction for controlling speech style.\n * Use natural language to describe the desired speaking style,\n * pace, tone, accent, or other characteristics.\n *\n * With Gemini 3.1 Flash TTS, you can also use inline audio tags like\n * `[whispering]`, `[laughs]`, `[excited]` directly in the input text\n * to control delivery.\n *\n * @example \"Speak slowly and calmly, as if telling a bedtime story\"\n * @example \"Use an upbeat, enthusiastic tone with moderate pace\"\n * @example \"Speak with a British accent\"\n */\n systemInstruction?: string\n\n /**\n * Language code hint for the speech synthesis.\n * Gemini 3.1 Flash TTS supports 70+ languages with auto-detection;\n * the 2.5 TTS models support 24 languages.\n *\n * @example \"en-US\" for American English\n * @example \"es-ES\" for Spanish (Spain)\n * @example \"ja-JP\" for Japanese\n */\n languageCode?: string\n}\n\n/**\n * Configuration for Gemini TTS adapter\n *\n * @experimental Gemini TTS is an experimental feature.\n */\nexport interface GeminiTTSConfig extends GeminiClientConfig {}\n\n/** Model type for Gemini TTS */\nexport type GeminiTTSModel = (typeof GEMINI_TTS_MODELS)[number]\n\n/**\n * Gemini Text-to-Speech Adapter\n *\n * Tree-shakeable adapter for Gemini TTS functionality.\n *\n * **IMPORTANT**: Gemini TTS uses the Live API (WebSocket-based) which requires\n * different handling than traditional REST APIs. This adapter provides a\n * simplified interface but may have limitations.\n *\n * @experimental Gemini TTS is an experimental feature and may change.\n *\n * Models:\n * - gemini-2.5-flash-preview-tts\n */\nexport class GeminiTTSAdapter<\n TModel extends GeminiTTSModel,\n> extends BaseTTSAdapter<TModel, GeminiTTSProviderOptions> {\n readonly name = 'gemini' as const\n\n private readonly client: GoogleGenAI\n\n constructor(config: GeminiTTSConfig, model: TModel) {\n super(model, config)\n this.client = createGeminiClient(config)\n }\n\n /**\n * Generate speech from text using Gemini's TTS model.\n *\n * @experimental This implementation is experimental and may change.\n * @see https://ai.google.dev/gemini-api/docs/speech-generation\n */\n async generateSpeech(\n options: TTSOptions<GeminiTTSProviderOptions>,\n ): Promise<TTSResult> {\n const { model, text, modelOptions, voice, logger } = options\n\n logger.request(`activity=generateSpeech provider=gemini model=${model}`, {\n provider: 'gemini',\n model,\n })\n\n const speechConfig: SpeechConfig = {}\n\n if (modelOptions?.multiSpeakerVoiceConfig) {\n // Validate multi-speaker config: 1 or 2 speakers allowed.\n const speakerConfigs =\n modelOptions.multiSpeakerVoiceConfig.speakerVoiceConfigs\n if (\n !Array.isArray(speakerConfigs) ||\n speakerConfigs.length < 1 ||\n speakerConfigs.length > 2\n ) {\n throw new Error(\n `Gemini TTS multiSpeakerVoiceConfig.speakerVoiceConfigs must contain 1 or 2 speakers; received ${Array.isArray(speakerConfigs) ? speakerConfigs.length : 'non-array'}.`,\n )\n }\n speechConfig.multiSpeakerVoiceConfig =\n modelOptions.multiSpeakerVoiceConfig\n } else {\n // Honor the standard TTSOptions.voice (used by every other TTS adapter)\n // as a fallback for the prebuilt voice name. If an explicit\n // modelOptions.voiceConfig is supplied its values win — but we still\n // fall back to `voice` / 'Kore' if the supplied voiceConfig is missing\n // prebuiltVoiceConfig.voiceName.\n if (\n voice !== undefined &&\n !(GEMINI_TTS_VOICES as ReadonlyArray<string>).includes(voice)\n ) {\n throw new Error(\n `Invalid Gemini TTS voice \"${voice}\". Valid voices are: ${GEMINI_TTS_VOICES.join(', ')}.`,\n )\n }\n const defaultVoiceName = (voice as GeminiTTSVoice | undefined) ?? 'Kore'\n const supplied = modelOptions?.voiceConfig\n const resolvedVoiceName =\n supplied?.prebuiltVoiceConfig?.voiceName ?? defaultVoiceName\n speechConfig.voiceConfig = {\n prebuiltVoiceConfig: { voiceName: resolvedVoiceName },\n }\n }\n\n if (modelOptions?.languageCode) {\n speechConfig.languageCode = modelOptions.languageCode\n }\n\n try {\n const response = await this.client.models.generateContent({\n model,\n contents: [\n {\n role: 'user',\n parts: [{ text }],\n },\n ],\n config: {\n responseModalities: ['AUDIO'],\n speechConfig,\n // systemInstruction belongs inside `config` per the @google/genai\n // contract — matches sibling Gemini adapters (summarize, text).\n ...(modelOptions?.systemInstruction && {\n systemInstruction: modelOptions.systemInstruction,\n }),\n },\n })\n\n // Extract audio data from response\n const candidate = response.candidates?.[0]\n const parts = candidate?.content?.parts\n\n if (!parts || parts.length === 0) {\n throw new Error('No audio output received from Gemini TTS')\n }\n\n // Look for inline data (audio)\n const audioPart = parts.find((part: any) =>\n part.inlineData?.mimeType?.startsWith('audio/'),\n )\n\n if (!audioPart || !audioPart.inlineData || !audioPart.inlineData.data) {\n throw new Error('No audio data in Gemini TTS response')\n }\n\n const audioBase64 = audioPart.inlineData.data\n // mime is guaranteed by the `startsWith('audio/')` find predicate above.\n const mimeType = audioPart.inlineData.mimeType as string\n\n // Surface token usage (with per-modality breakdown) when Gemini reports\n // it. Spread conditionally for exactOptionalPropertyTypes — shared by both\n // the PCM→WAV and pass-through return paths below.\n const usageField = response.usageMetadata\n ? { usage: buildGeminiUsage(response.usageMetadata) }\n : {}\n\n // Gemini TTS models return raw 16-bit LE PCM with a mime type like\n // `audio/L16;codec=pcm;rate=24000`. That isn't playable in an <audio>\n // element and the bare string isn't a usable file extension, so we\n // prepend a RIFF/WAV header here and normalize the result to audio/wav.\n const pcm = parsePcmMimeType(mimeType)\n if (pcm) {\n const wavBase64 = wrapPcmBase64AsWav(\n audioBase64,\n pcm.sampleRate,\n pcm.channels,\n pcm.bitsPerSample,\n )\n return {\n id: generateId(this.name),\n model,\n audio: wavBase64,\n format: 'wav',\n contentType: 'audio/wav',\n ...usageField,\n }\n }\n\n // Strip any mime parameters (e.g. `audio/ogg;codec=opus`) before pulling\n // the subtype out as the file format. `String.split` always returns at\n // least one element, so [0] is defined.\n const format = (mimeType.split(';')[0] ?? '').split('/')[1] || 'wav'\n\n return {\n id: generateId(this.name),\n model,\n audio: audioBase64,\n format,\n contentType: mimeType,\n ...usageField,\n }\n } catch (error) {\n logger.errors('gemini.generateSpeech fatal', {\n error,\n source: 'gemini.generateSpeech',\n })\n throw error\n }\n }\n}\n\nfunction parsePcmMimeType(\n mimeType: string,\n): { sampleRate: number; channels: number; bitsPerSample: number } | undefined {\n const normalized = mimeType.toLowerCase()\n const subtype = (normalized.split(';')[0] ?? '').split('/')[1] ?? ''\n // Exclude containerized wav (e.g. `audio/wav;codec=pcm`) — those already\n // carry a RIFF header and must not be re-wrapped.\n if (subtype.includes('wav')) return undefined\n\n // Accept the variants Gemini and other providers actually emit:\n // - audio/L16;codec=pcm;rate=24000 (IANA PCM with bit depth in the type)\n // - audio/L24 and friends\n // - audio/pcm and audio/x-pcm\n // - anything else that explicitly tags codec=pcm and isn't wav-containered\n const bitDepthMatch = /^audio\\/l(\\d+)/.exec(normalized)\n const isPcm =\n bitDepthMatch !== null ||\n normalized.startsWith('audio/pcm') ||\n normalized.startsWith('audio/x-pcm') ||\n normalized.includes('codec=pcm')\n if (!isPcm) return undefined\n\n const rateMatch = /rate=(\\d+)/.exec(normalized)\n const channelsMatch = /channels=(\\d+)/.exec(normalized)\n // Default to 16-bit when the mime type doesn't specify — matches Gemini's\n // audio/L16;codec=pcm;rate=24000 response.\n const bitsPerSample = bitDepthMatch ? Number(bitDepthMatch[1]) : 16\n return {\n sampleRate: rateMatch ? Number(rateMatch[1]) : 24000,\n channels: channelsMatch ? Number(channelsMatch[1]) : 1,\n bitsPerSample,\n }\n}\n\nfunction wrapPcmBase64AsWav(\n pcmBase64: string,\n sampleRate: number,\n channels = 1,\n bitsPerSample = 16,\n): string {\n // The WAV writer below emits a 16-bit PCM fmt chunk. If the source claims a\n // different bit depth we'd be lying about the payload, so bail out loudly\n // rather than producing a corrupt file.\n if (bitsPerSample !== 16) {\n throw new Error(\n `Unsupported PCM bit depth ${bitsPerSample}: only 16-bit PCM can be wrapped as WAV.`,\n )\n }\n\n const pcmBytes =\n typeof Buffer !== 'undefined'\n ? new Uint8Array(Buffer.from(pcmBase64, 'base64'))\n : decodeBase64(pcmBase64)\n\n const byteRate = (sampleRate * channels * bitsPerSample) / 8\n const blockAlign = (channels * bitsPerSample) / 8\n const dataSize = pcmBytes.byteLength\n const buffer = new ArrayBuffer(44 + dataSize)\n const view = new DataView(buffer)\n\n writeAscii(view, 0, 'RIFF')\n view.setUint32(4, 36 + dataSize, true)\n writeAscii(view, 8, 'WAVE')\n writeAscii(view, 12, 'fmt ')\n view.setUint32(16, 16, true)\n view.setUint16(20, 1, true)\n view.setUint16(22, channels, true)\n view.setUint32(24, sampleRate, true)\n view.setUint32(28, byteRate, true)\n view.setUint16(32, blockAlign, true)\n view.setUint16(34, bitsPerSample, true)\n writeAscii(view, 36, 'data')\n view.setUint32(40, dataSize, true)\n new Uint8Array(buffer, 44).set(pcmBytes)\n\n if (typeof Buffer !== 'undefined') {\n return Buffer.from(buffer).toString('base64')\n }\n let binary = ''\n const bytes = new Uint8Array(buffer)\n for (const byte of bytes) {\n binary += String.fromCharCode(byte)\n }\n return btoa(binary)\n}\n\nfunction decodeBase64(b64: string): Uint8Array {\n const binary = atob(b64)\n const out = new Uint8Array(binary.length)\n for (let i = 0; i < binary.length; i += 1) out[i] = binary.charCodeAt(i)\n return out\n}\n\nfunction writeAscii(view: DataView, offset: number, text: string): void {\n for (let i = 0; i < text.length; i += 1) {\n view.setUint8(offset + i, text.charCodeAt(i))\n }\n}\n\n/**\n * Creates a Gemini TTS adapter with explicit API key.\n * Type resolution happens here at the call site.\n *\n * @experimental Gemini TTS is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'gemini-2.5-flash-preview-tts')\n * @param apiKey - Your Google API key\n * @param config - Optional additional configuration\n * @returns Configured Gemini TTS adapter instance with resolved types\n *\n * @example\n * ```typescript\n * const adapter = createGeminiSpeech('gemini-2.5-flash-preview-tts', \"your-api-key\");\n *\n * const result = await generateSpeech({\n * adapter,\n * text: 'Hello, world!'\n * });\n * ```\n */\nexport function createGeminiSpeech<TModel extends GeminiTTSModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GeminiTTSConfig, 'apiKey'>,\n): GeminiTTSAdapter<TModel> {\n // Put apiKey LAST so caller-supplied config can't silently override the\n // explicit argument.\n return new GeminiTTSAdapter({ ...config, apiKey }, model)\n}\n\n/**\n * Creates a Gemini speech adapter with automatic API key detection from environment variables.\n * Type resolution happens here at the call site.\n *\n * @experimental Gemini TTS is an experimental feature and may change.\n *\n * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (Browser with injected env)\n *\n * @param model - The model name (e.g., 'gemini-2.5-flash-preview-tts')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Gemini speech adapter instance with resolved types\n * @throws Error if GOOGLE_API_KEY or GEMINI_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * // Automatically uses GOOGLE_API_KEY from environment\n * const adapter = geminiSpeech('gemini-2.5-flash-preview-tts');\n *\n * const result = await generateSpeech({\n * adapter,\n * text: 'Welcome to TanStack AI!'\n * });\n * ```\n */\nexport function geminiSpeech<TModel extends GeminiTTSModel>(\n model: TModel,\n config?: Omit<GeminiTTSConfig, 'apiKey'>,\n): GeminiTTSAdapter<TModel> {\n const apiKey = getGeminiApiKeyFromEnv()\n return createGeminiSpeech(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;;AAuHO,MAAM,yBAEH,eAAiD;AAAA,EAChD,OAAO;AAAA,EAEC;AAAA,EAEjB,YAAY,QAAyB,OAAe;AAClD,UAAM,OAAO,MAAM;AACnB,SAAK,SAAS,mBAAmB,MAAM;AAAA,EACzC;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAQA,MAAM,eACJ,SACoB;AACpB,UAAM,EAAE,OAAO,MAAM,cAAc,OAAO,WAAW;AAErD,WAAO,QAAQ,iDAAiD,KAAK,IAAI;AAAA,MACvE,UAAU;AAAA,MACV;AAAA,IAAA,CACD;AAED,UAAM,eAA6B,CAAA;AAEnC,QAAI,cAAc,yBAAyB;AAEzC,YAAM,iBACJ,aAAa,wBAAwB;AACvC,UACE,CAAC,MAAM,QAAQ,cAAc,KAC7B,eAAe,SAAS,KACxB,eAAe,SAAS,GACxB;AACA,cAAM,IAAI;AAAA,UACR,iGAAiG,MAAM,QAAQ,cAAc,IAAI,eAAe,SAAS,WAAW;AAAA,QAAA;AAAA,MAExK;AACA,mBAAa,0BACX,aAAa;AAAA,IACjB,OAAO;AAML,UACE,UAAU,UACV,CAAE,kBAA4C,SAAS,KAAK,GAC5D;AACA,cAAM,IAAI;AAAA,UACR,6BAA6B,KAAK,wBAAwB,kBAAkB,KAAK,IAAI,CAAC;AAAA,QAAA;AAAA,MAE1F;AACA,YAAM,mBAAoB,SAAwC;AAClE,YAAM,WAAW,cAAc;AAC/B,YAAM,oBACJ,UAAU,qBAAqB,aAAa;AAC9C,mBAAa,cAAc;AAAA,QACzB,qBAAqB,EAAE,WAAW,kBAAA;AAAA,MAAkB;AAAA,IAExD;AAEA,QAAI,cAAc,cAAc;AAC9B,mBAAa,eAAe,aAAa;AAAA,IAC3C;AAEA,QAAI;AACF,YAAM,WAAW,MAAM,KAAK,OAAO,OAAO,gBAAgB;AAAA,QACxD;AAAA,QACA,UAAU;AAAA,UACR;AAAA,YACE,MAAM;AAAA,YACN,OAAO,CAAC,EAAE,KAAA,CAAM;AAAA,UAAA;AAAA,QAClB;AAAA,QAEF,QAAQ;AAAA,UACN,oBAAoB,CAAC,OAAO;AAAA,UAC5B;AAAA;AAAA;AAAA,UAGA,GAAI,cAAc,qBAAqB;AAAA,YACrC,mBAAmB,aAAa;AAAA,UAAA;AAAA,QAClC;AAAA,MACF,CACD;AAGD,YAAM,YAAY,SAAS,aAAa,CAAC;AACzC,YAAM,QAAQ,WAAW,SAAS;AAElC,UAAI,CAAC,SAAS,MAAM,WAAW,GAAG;AAChC,cAAM,IAAI,MAAM,0CAA0C;AAAA,MAC5D;AAGA,YAAM,YAAY,MAAM;AAAA,QAAK,CAAC,SAC5B,KAAK,YAAY,UAAU,WAAW,QAAQ;AAAA,MAAA;AAGhD,UAAI,CAAC,aAAa,CAAC,UAAU,cAAc,CAAC,UAAU,WAAW,MAAM;AACrE,cAAM,IAAI,MAAM,sCAAsC;AAAA,MACxD;AAEA,YAAM,cAAc,UAAU,WAAW;AAEzC,YAAM,WAAW,UAAU,WAAW;AAKtC,YAAM,aAAa,SAAS,gBACxB,EAAE,OAAO,iBAAiB,SAAS,aAAa,EAAA,IAChD,CAAA;AAMJ,YAAM,MAAM,iBAAiB,QAAQ;AACrC,UAAI,KAAK;AACP,cAAM,YAAY;AAAA,UAChB;AAAA,UACA,IAAI;AAAA,UACJ,IAAI;AAAA,UACJ,IAAI;AAAA,QAAA;AAEN,eAAO;AAAA,UACL,IAAI,WAAW,KAAK,IAAI;AAAA,UACxB;AAAA,UACA,OAAO;AAAA,UACP,QAAQ;AAAA,UACR,aAAa;AAAA,UACb,GAAG;AAAA,QAAA;AAAA,MAEP;AAKA,YAAM,UAAU,SAAS,MAAM,GAAG,EAAE,CAAC,KAAK,IAAI,MAAM,GAAG,EAAE,CAAC,KAAK;AAE/D,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA,OAAO;AAAA,QACP;AAAA,QACA,aAAa;AAAA,QACb,GAAG;AAAA,MAAA;AAAA,IAEP,SAAS,OAAO;AACd,aAAO,OAAO,+BAA+B;AAAA,QAC3C;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AAEA,SAAS,iBACP,UAC6E;AAC7E,QAAM,aAAa,SAAS,YAAA;AAC5B,QAAM,WAAW,WAAW,MAAM,GAAG,EAAE,CAAC,KAAK,IAAI,MAAM,GAAG,EAAE,CAAC,KAAK;AAGlE,MAAI,QAAQ,SAAS,KAAK,EAAG,QAAO;AAOpC,QAAM,gBAAgB,iBAAiB,KAAK,UAAU;AACtD,QAAM,QACJ,kBAAkB,QAClB,WAAW,WAAW,WAAW,KACjC,WAAW,WAAW,aAAa,KACnC,WAAW,SAAS,WAAW;AACjC,MAAI,CAAC,MAAO,QAAO;AAEnB,QAAM,YAAY,aAAa,KAAK,UAAU;AAC9C,QAAM,gBAAgB,iBAAiB,KAAK,UAAU;AAGtD,QAAM,gBAAgB,gBAAgB,OAAO,cAAc,CAAC,CAAC,IAAI;AACjE,SAAO;AAAA,IACL,YAAY,YAAY,OAAO,UAAU,CAAC,CAAC,IAAI;AAAA,IAC/C,UAAU,gBAAgB,OAAO,cAAc,CAAC,CAAC,IAAI;AAAA,IACrD;AAAA,EAAA;AAEJ;AAEA,SAAS,mBACP,WACA,YACA,WAAW,GACX,gBAAgB,IACR;AAIR,MAAI,kBAAkB,IAAI;AACxB,UAAM,IAAI;AAAA,MACR,6BAA6B,aAAa;AAAA,IAAA;AAAA,EAE9C;AAEA,QAAM,WACJ,OAAO,WAAW,cACd,IAAI,WAAW,OAAO,KAAK,WAAW,QAAQ,CAAC,IAC/C,aAAa,SAAS;AAE5B,QAAM,WAAY,aAAa,WAAW,gBAAiB;AAC3D,QAAM,aAAc,WAAW,gBAAiB;AAChD,QAAM,WAAW,SAAS;AAC1B,QAAM,SAAS,IAAI,YAAY,KAAK,QAAQ;AAC5C,QAAM,OAAO,IAAI,SAAS,MAAM;AAEhC,aAAW,MAAM,GAAG,MAAM;AAC1B,OAAK,UAAU,GAAG,KAAK,UAAU,IAAI;AACrC,aAAW,MAAM,GAAG,MAAM;AAC1B,aAAW,MAAM,IAAI,MAAM;AAC3B,OAAK,UAAU,IAAI,IAAI,IAAI;AAC3B,OAAK,UAAU,IAAI,GAAG,IAAI;AAC1B,OAAK,UAAU,IAAI,UAAU,IAAI;AACjC,OAAK,UAAU,IAAI,YAAY,IAAI;AACnC,OAAK,UAAU,IAAI,UAAU,IAAI;AACjC,OAAK,UAAU,IAAI,YAAY,IAAI;AACnC,OAAK,UAAU,IAAI,eAAe,IAAI;AACtC,aAAW,MAAM,IAAI,MAAM;AAC3B,OAAK,UAAU,IAAI,UAAU,IAAI;AACjC,MAAI,WAAW,QAAQ,EAAE,EAAE,IAAI,QAAQ;AAEvC,MAAI,OAAO,WAAW,aAAa;AACjC,WAAO,OAAO,KAAK,MAAM,EAAE,SAAS,QAAQ;AAAA,EAC9C;AACA,MAAI,SAAS;AACb,QAAM,QAAQ,IAAI,WAAW,MAAM;AACnC,aAAW,QAAQ,OAAO;AACxB,cAAU,OAAO,aAAa,IAAI;AAAA,EACpC;AACA,SAAO,KAAK,MAAM;AACpB;AAEA,SAAS,aAAa,KAAyB;AAC7C,QAAM,SAAS,KAAK,GAAG;AACvB,QAAM,MAAM,IAAI,WAAW,OAAO,MAAM;AACxC,WAAS,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK,EAAG,KAAI,CAAC,IAAI,OAAO,WAAW,CAAC;AACvE,SAAO;AACT;AAEA,SAAS,WAAW,MAAgB,QAAgB,MAAoB;AACtE,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK,GAAG;AACvC,SAAK,SAAS,SAAS,GAAG,KAAK,WAAW,CAAC,CAAC;AAAA,EAC9C;AACF;AAuBO,SAAS,mBACd,OACA,QACA,QAC0B;AAG1B,SAAO,IAAI,iBAAiB,EAAE,GAAG,QAAQ,OAAA,GAAU,KAAK;AAC1D;AA4BO,SAAS,aACd,OACA,QAC0B;AAC1B,QAAM,SAAS,uBAAA;AACf,SAAO,mBAAmB,OAAO,QAAQ,MAAM;AACjD;"}
@@ -1,47 +1,92 @@
1
1
  import { BaseVideoAdapter, DurationOptions } from '@tanstack/ai/adapters';
2
2
  import { VideoGenerationOptions, VideoJobResult, VideoStatusResult, VideoUrlResult } from '@tanstack/ai';
3
3
  import { GoogleGenAI } from '@google/genai';
4
- import { GeminiVideoModel, GeminiVideoModelDurationByName, GeminiVideoModelInputModalitiesByName, GeminiVideoModelProviderOptionsByName, GeminiVideoModelSizeByName, GeminiVideoProviderOptions, GeminiVideoSize } from '../video/video-provider-options.js';
5
- import { GeminiClientConfig } from '../utils.js';
4
+ import { GeminiVideoModel, GeminiVideoModelDurationByName, GeminiVideoModelInputModalitiesByName, GeminiVideoModelProviderOptionsByName, GeminiVideoModelSizeByName, GeminiVideoSize } from '../video/video-provider-options.js';
5
+ import { GeminiClientConfig } from '../utils/client.js';
6
6
  /**
7
7
  * Configuration for Gemini video adapter.
8
8
  *
9
9
  * @experimental Video generation is an experimental feature and may change.
10
10
  */
11
11
  export interface GeminiVideoConfig extends GeminiClientConfig {
12
+ /**
13
+ * Opt into fetching HTTP(S) image URL inputs. Veo's predict API accepts
14
+ * only inline `imageBytes` or a `gcsUri`, so an HTTP(S) URL has to be
15
+ * downloaded and base64-encoded locally — which buffers the whole image in
16
+ * memory and can OOM constrained runtimes (e.g. Cloudflare Workers). When
17
+ * `false` (the default), HTTP(S) URL image inputs throw; pass a `data:` URI
18
+ * or a `gs://` reference, or set this to `true` to opt into buffering.
19
+ */
20
+ allowUrlFetch?: boolean;
12
21
  }
13
22
  /**
14
- * Gemini Veo Video Generation Adapter
23
+ * Gemini Video Generation Adapter (Veo + Gemini Omni Flash)
15
24
  *
16
- * Tree-shakeable adapter for Google Veo video generation. Veo runs as a
17
- * long-running operation: `createVideoJob` starts the operation via the
18
- * `:predictLongRunning` endpoint, `getVideoStatus` polls it, and
19
- * `getVideoUrl` extracts the generated video's URI once it completes.
25
+ * Tree-shakeable adapter for Google video generation, routing by model:
20
26
  *
21
- * Image prompt parts are routed by `metadata.role`:
27
+ * **Veo models** run as a long-running operation: `createVideoJob` starts
28
+ * the operation via the `:predictLongRunning` endpoint, `getVideoStatus`
29
+ * polls it, and `getVideoUrl` extracts the generated video's URI once it
30
+ * completes. Image prompt parts are routed by `metadata.role`:
22
31
  * - `'start_frame'` (or the first un-roled image) → the input image the
23
32
  * video starts from
24
33
  * - `'end_frame'` → `lastFrame` (the frame the video ends on)
25
34
  * - `'reference'` / `'character'` → `referenceImages` (asset references,
26
35
  * Veo 3.1)
27
36
  *
28
- * Note: the returned video URI is served by the Gemini Files API and
37
+ * Note: the returned Veo video URI is served by the Gemini Files API and
29
38
  * requires the API key (`x-goog-api-key` header or `?key=` query
30
39
  * parameter) to download.
31
40
  *
41
+ * **Gemini Omni Flash** (`gemini-omni-flash-preview`) only serves the
42
+ * Interactions API: `createVideoJob` creates a background interaction with
43
+ * `response_modalities: ['video']`, `getVideoStatus` polls it by id, and
44
+ * `getVideoUrl` returns the inline base64 MP4 as a `data:` URL (or the
45
+ * Files API URI when the server delivers by reference). Image and video
46
+ * prompt parts are sent as interaction content blocks, grouped as images,
47
+ * then videos, then the text prompt (interleaving is not preserved); pass
48
+ * `modelOptions.previous_interaction_id` to conversationally edit a prior
49
+ * Omni generation.
50
+ *
32
51
  * @experimental Video generation is an experimental feature and may change.
33
52
  */
34
- export declare class GeminiVideoAdapter<TModel extends GeminiVideoModel> extends BaseVideoAdapter<TModel, GeminiVideoProviderOptions, GeminiVideoModelProviderOptionsByName, GeminiVideoModelSizeByName, GeminiVideoModelInputModalitiesByName, GeminiVideoModelDurationByName> {
53
+ export declare class GeminiVideoAdapter<TModel extends GeminiVideoModel> extends BaseVideoAdapter<TModel, GeminiVideoModelProviderOptionsByName[TModel], GeminiVideoModelProviderOptionsByName, GeminiVideoModelSizeByName, GeminiVideoModelInputModalitiesByName, GeminiVideoModelDurationByName> {
35
54
  readonly name: "gemini";
36
55
  protected client: GoogleGenAI;
56
+ private readonly allowUrlFetch;
37
57
  constructor(config: GeminiVideoConfig, model: TModel);
38
- createVideoJob(options: VideoGenerationOptions<GeminiVideoProviderOptions, GeminiVideoSize, GeminiVideoModelDurationByName[TModel]>): Promise<VideoJobResult>;
58
+ createVideoJob(options: VideoGenerationOptions<GeminiVideoModelProviderOptionsByName[TModel], GeminiVideoSize, GeminiVideoModelDurationByName[TModel]>): Promise<VideoJobResult>;
59
+ /**
60
+ * Gemini Omni Flash job creation via the Interactions API. Creates a
61
+ * background interaction requesting video output; the interaction id is
62
+ * the job id polled by `getVideoStatus` / `getVideoUrl`.
63
+ */
64
+ private createInteractionsVideoJob;
39
65
  /**
40
66
  * Route image prompt parts onto Veo's request fields by `metadata.role`.
41
67
  */
42
68
  private routeImageParts;
43
69
  getVideoStatus(jobId: string): Promise<VideoStatusResult>;
70
+ /**
71
+ * Poll an Omni background interaction. `in_progress` maps to
72
+ * 'processing'; a `completed` interaction with no video content (e.g.
73
+ * filtered output) is surfaced as a failure so `getVideoUrl` doesn't
74
+ * throw on an empty response. `requires_action` also fails: the adapter
75
+ * never sends tools, so it can only arise via
76
+ * `previous_interaction_id` chaining onto a tool-bearing interaction —
77
+ * and such an interaction never progresses without a client response,
78
+ * so polling it would spin until timeout.
79
+ */
80
+ private getInteractionsVideoStatus;
44
81
  getVideoUrl(jobId: string): Promise<VideoUrlResult>;
82
+ /**
83
+ * Extract the finished Omni video. Inline base64 output (the API default)
84
+ * becomes a `data:` URL — matching the OpenAI Sora adapter's inline
85
+ * delivery — and URI delivery passes through (Files API URIs need the API
86
+ * key to download, like Veo). Usage carries the video-modality output
87
+ * tokens (Omni bills per second of video, reported as tokens).
88
+ */
89
+ private getInteractionsVideoUrl;
45
90
  availableDurations(): DurationOptions<GeminiVideoModelDurationByName[TModel]>;
46
91
  snapDuration(seconds: number): GeminiVideoModelDurationByName[TModel] | undefined;
47
92
  /**
@@ -51,6 +96,10 @@ export declare class GeminiVideoAdapter<TModel extends GeminiVideoModel> extends
51
96
  * the job ID rather than passing an object literal.
52
97
  */
53
98
  private getOperation;
99
+ /**
100
+ * Fetch an Omni background interaction by id.
101
+ */
102
+ private getInteraction;
54
103
  }
55
104
  /**
56
105
  * Creates a Gemini video adapter with an explicit API key.
@@ -3,14 +3,14 @@ import { resolveMediaPrompt } from "@tanstack/ai";
3
3
  import { BaseVideoAdapter, snapToDurationOption } from "@tanstack/ai/adapters";
4
4
  import { arrayBufferToBase64 } from "@tanstack/ai-utils";
5
5
  import { createGeminiClient, getGeminiApiKeyFromEnv } from "../utils/client.js";
6
- import { getGeminiVideoDurationOptions } from "../video/video-provider-options.js";
6
+ import { isInteractionsVideoModel, getGeminiVideoDurationOptions } from "../video/video-provider-options.js";
7
7
  function operationErrorMessage(error) {
8
8
  if (typeof error.message === "string" && error.message.length > 0) {
9
9
  return error.message;
10
10
  }
11
11
  return JSON.stringify(error);
12
12
  }
13
- async function imagePartToVeoImage(part) {
13
+ async function imagePartToVeoImage(part, allowUrlFetch) {
14
14
  if (part.source.type === "data") {
15
15
  return {
16
16
  imageBytes: part.source.value,
@@ -36,6 +36,11 @@ async function imagePartToVeoImage(part) {
36
36
  mimeType: match[1] || part.source.mimeType || "image/png"
37
37
  };
38
38
  }
39
+ if (!allowUrlFetch) {
40
+ throw new Error(
41
+ `gemini Veo: HTTP(S) URL image inputs are not fetched by default because Veo accepts only inline bytes, so the image would be downloaded and buffered in memory (risking OOM on constrained runtimes). Pass a data: URI or a gs:// reference, or set \`allowUrlFetch: true\` on the adapter config to opt into fetching. URL: ${url}`
42
+ );
43
+ }
39
44
  const response = await fetch(url);
40
45
  if (!response.ok) {
41
46
  throw new Error(
@@ -49,19 +54,70 @@ async function imagePartToVeoImage(part) {
49
54
  mimeType: part.source.mimeType || blob.type || "image/png"
50
55
  };
51
56
  }
57
+ function mediaPartToInteractionsContent(part) {
58
+ const mimeType = part.source.mimeType;
59
+ if (part.type === "image") {
60
+ return part.source.type === "data" ? { type: "image", data: part.source.value, mime_type: mimeType } : { type: "image", uri: part.source.value, mime_type: mimeType };
61
+ }
62
+ return part.source.type === "data" ? { type: "video", data: part.source.value, mime_type: mimeType } : { type: "video", uri: part.source.value, mime_type: mimeType };
63
+ }
64
+ function extractInteractionVideo(interaction) {
65
+ const direct = interaction.output_video;
66
+ if (direct && (direct.data || direct.uri)) {
67
+ return {
68
+ data: direct.data,
69
+ uri: direct.uri,
70
+ mimeType: direct.mime_type || "video/mp4"
71
+ };
72
+ }
73
+ const steps = interaction.steps ?? [];
74
+ for (let i = steps.length - 1; i >= 0; i--) {
75
+ const step = steps[i];
76
+ if (step?.type !== "model_output") continue;
77
+ for (const block of step.content ?? []) {
78
+ if (block.type === "video" && (block.data || block.uri)) {
79
+ return {
80
+ data: block.data,
81
+ uri: block.uri,
82
+ mimeType: block.mime_type || "video/mp4"
83
+ };
84
+ }
85
+ }
86
+ }
87
+ return void 0;
88
+ }
89
+ function interactionUsageToTokenUsage(usage) {
90
+ if (!usage) return void 0;
91
+ const videoTokens = usage.output_tokens_by_modality?.find(
92
+ (entry) => entry.modality === "video"
93
+ )?.tokens;
94
+ const promptTokens = usage.total_input_tokens ?? 0;
95
+ const completionTokens = usage.total_output_tokens ?? videoTokens ?? 0;
96
+ return {
97
+ promptTokens,
98
+ completionTokens,
99
+ totalTokens: usage.total_tokens ?? promptTokens + completionTokens
100
+ };
101
+ }
52
102
  class GeminiVideoAdapter extends BaseVideoAdapter {
53
103
  name = "gemini";
54
104
  client;
105
+ allowUrlFetch;
55
106
  constructor(config, model) {
56
107
  super({}, model);
57
108
  this.client = createGeminiClient(config);
109
+ this.allowUrlFetch = config.allowUrlFetch ?? false;
58
110
  }
59
111
  async createVideoJob(options) {
60
- const { prompt, size, duration, modelOptions, logger } = options;
112
+ const { prompt, size, duration, logger } = options;
61
113
  logger.request(
62
114
  `activity=video.create provider=${this.name} model=${this.model} size=${size ?? "default"} duration=${duration ?? "default"}`,
63
115
  { provider: this.name, model: this.model }
64
116
  );
117
+ if (isInteractionsVideoModel(this.model)) {
118
+ return await this.createInteractionsVideoJob(options);
119
+ }
120
+ const modelOptions = options.modelOptions;
65
121
  try {
66
122
  const resolved = resolveMediaPrompt(prompt);
67
123
  if (resolved.videos.length > 0) {
@@ -104,6 +160,68 @@ class GeminiVideoAdapter extends BaseVideoAdapter {
104
160
  throw error;
105
161
  }
106
162
  }
163
+ /**
164
+ * Gemini Omni Flash job creation via the Interactions API. Creates a
165
+ * background interaction requesting video output; the interaction id is
166
+ * the job id polled by `getVideoStatus` / `getVideoUrl`.
167
+ */
168
+ async createInteractionsVideoJob(options) {
169
+ const { prompt, size, duration, logger } = options;
170
+ const modelOptions = options.modelOptions;
171
+ try {
172
+ const resolved = resolveMediaPrompt(prompt);
173
+ if (resolved.audios.length > 0) {
174
+ throw new Error(
175
+ `${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`
176
+ );
177
+ }
178
+ const content = [
179
+ ...resolved.images.map(mediaPartToInteractionsContent),
180
+ ...resolved.videos.map(mediaPartToInteractionsContent)
181
+ ];
182
+ if (resolved.text) {
183
+ content.push({ type: "text", text: resolved.text });
184
+ }
185
+ if (content.length === 0) {
186
+ throw new Error(
187
+ `${this.name}.createVideoJob: the prompt produced no content to send (model: ${this.model}).`
188
+ );
189
+ }
190
+ const durations = this.availableDurations();
191
+ if (duration !== void 0 && durations.kind === "range" && (duration < durations.min || duration > durations.max)) {
192
+ throw new Error(
193
+ `${this.name}.createVideoJob: duration ${duration}s is outside the ${durations.min}–${durations.max}s range supported by ${this.model}. Use snapDuration() to snap arbitrary values into range.`
194
+ );
195
+ }
196
+ const responseFormat = size !== void 0 || duration !== void 0 ? {
197
+ response_format: {
198
+ type: "video",
199
+ ...size !== void 0 && { aspect_ratio: size },
200
+ ...duration !== void 0 && { duration: `${duration}s` }
201
+ }
202
+ } : {};
203
+ const interaction = await this.client.interactions.create({
204
+ ...modelOptions,
205
+ model: this.model,
206
+ input: [{ type: "user_input", content }],
207
+ response_modalities: ["video"],
208
+ background: true,
209
+ ...responseFormat
210
+ });
211
+ if (!interaction.id) {
212
+ throw new Error(
213
+ "Gemini Omni did not return an interaction id for the video generation job."
214
+ );
215
+ }
216
+ return { jobId: interaction.id, model: this.model };
217
+ } catch (error) {
218
+ logger.errors(`${this.name}.createVideoJob fatal`, {
219
+ error,
220
+ source: `${this.name}.createVideoJob`
221
+ });
222
+ throw error;
223
+ }
224
+ }
107
225
  /**
108
226
  * Route image prompt parts onto Veo's request fields by `metadata.role`.
109
227
  */
@@ -120,13 +238,13 @@ class GeminiVideoAdapter extends BaseVideoAdapter {
120
238
  `${this.name}: Veo accepts at most one 'end_frame' image.`
121
239
  );
122
240
  }
123
- lastFrame = await imagePartToVeoImage(part);
241
+ lastFrame = await imagePartToVeoImage(part, this.allowUrlFetch);
124
242
  break;
125
243
  }
126
244
  case "reference":
127
245
  case "character": {
128
246
  referenceImages.push({
129
- image: await imagePartToVeoImage(part),
247
+ image: await imagePartToVeoImage(part, this.allowUrlFetch),
130
248
  referenceType: VideoGenerationReferenceType.ASSET
131
249
  });
132
250
  break;
@@ -138,7 +256,7 @@ class GeminiVideoAdapter extends BaseVideoAdapter {
138
256
  `${this.name}: Veo accepts at most one starting image; received multiple 'start_frame'/un-roled images. Use metadata.role ('end_frame', 'reference') to disambiguate the others.`
139
257
  );
140
258
  }
141
- image = await imagePartToVeoImage(part);
259
+ image = await imagePartToVeoImage(part, this.allowUrlFetch);
142
260
  break;
143
261
  }
144
262
  case "mask":
@@ -151,6 +269,9 @@ class GeminiVideoAdapter extends BaseVideoAdapter {
151
269
  return { image, lastFrame, referenceImages };
152
270
  }
153
271
  async getVideoStatus(jobId) {
272
+ if (isInteractionsVideoModel(this.model)) {
273
+ return await this.getInteractionsVideoStatus(jobId);
274
+ }
154
275
  const operation = await this.getOperation(jobId);
155
276
  if (!operation.done) {
156
277
  return { jobId, status: "processing" };
@@ -173,7 +294,49 @@ class GeminiVideoAdapter extends BaseVideoAdapter {
173
294
  }
174
295
  return { jobId, status: "completed" };
175
296
  }
297
+ /**
298
+ * Poll an Omni background interaction. `in_progress` maps to
299
+ * 'processing'; a `completed` interaction with no video content (e.g.
300
+ * filtered output) is surfaced as a failure so `getVideoUrl` doesn't
301
+ * throw on an empty response. `requires_action` also fails: the adapter
302
+ * never sends tools, so it can only arise via
303
+ * `previous_interaction_id` chaining onto a tool-bearing interaction —
304
+ * and such an interaction never progresses without a client response,
305
+ * so polling it would spin until timeout.
306
+ */
307
+ async getInteractionsVideoStatus(jobId) {
308
+ const interaction = await this.getInteraction(jobId);
309
+ const status = interaction.status;
310
+ if (status === "in_progress") {
311
+ return { jobId, status: "processing" };
312
+ }
313
+ if (status === "requires_action") {
314
+ return {
315
+ jobId,
316
+ status: "failed",
317
+ error: "Gemini Omni interaction is waiting on a client action (tool response), which the video jobs flow does not support."
318
+ };
319
+ }
320
+ if (status === "completed") {
321
+ if (!extractInteractionVideo(interaction)) {
322
+ return {
323
+ jobId,
324
+ status: "failed",
325
+ error: "Gemini Omni completed the interaction without returning a video (the output may have been filtered)."
326
+ };
327
+ }
328
+ return { jobId, status: "completed" };
329
+ }
330
+ return {
331
+ jobId,
332
+ status: "failed",
333
+ error: `Gemini Omni video generation ended with status "${status}".`
334
+ };
335
+ }
176
336
  async getVideoUrl(jobId) {
337
+ if (isInteractionsVideoModel(this.model)) {
338
+ return await this.getInteractionsVideoUrl(jobId);
339
+ }
177
340
  const operation = await this.getOperation(jobId);
178
341
  if (!operation.done) {
179
342
  throw new Error(
@@ -194,6 +357,36 @@ class GeminiVideoAdapter extends BaseVideoAdapter {
194
357
  }
195
358
  return { jobId, url: uri };
196
359
  }
360
+ /**
361
+ * Extract the finished Omni video. Inline base64 output (the API default)
362
+ * becomes a `data:` URL — matching the OpenAI Sora adapter's inline
363
+ * delivery — and URI delivery passes through (Files API URIs need the API
364
+ * key to download, like Veo). Usage carries the video-modality output
365
+ * tokens (Omni bills per second of video, reported as tokens).
366
+ */
367
+ async getInteractionsVideoUrl(jobId) {
368
+ const interaction = await this.getInteraction(jobId);
369
+ const status = interaction.status;
370
+ if (status === "in_progress") {
371
+ throw new Error(
372
+ `Video is not ready yet. Check status first. Job ID: ${jobId}`
373
+ );
374
+ }
375
+ if (status !== "completed") {
376
+ throw new Error(
377
+ `Video generation failed: Gemini Omni interaction ended with status "${status}". Job ID: ${jobId}`
378
+ );
379
+ }
380
+ const video = extractInteractionVideo(interaction);
381
+ if (!video) {
382
+ throw new Error(
383
+ `Video not found in interaction response (the output may have been filtered). Job ID: ${jobId}`
384
+ );
385
+ }
386
+ const usage = interactionUsageToTokenUsage(interaction.usage);
387
+ const url = video.uri ?? `data:${video.mimeType};base64,${video.data}`;
388
+ return { jobId, url, ...usage && { usage } };
389
+ }
197
390
  availableDurations() {
198
391
  return getGeminiVideoDurationOptions(this.model);
199
392
  }
@@ -211,6 +404,12 @@ class GeminiVideoAdapter extends BaseVideoAdapter {
211
404
  operation.name = jobId;
212
405
  return await this.client.operations.getVideosOperation({ operation });
213
406
  }
407
+ /**
408
+ * Fetch an Omni background interaction by id.
409
+ */
410
+ async getInteraction(jobId) {
411
+ return await this.client.interactions.get(jobId);
412
+ }
214
413
  }
215
414
  function createGeminiVideo(model, apiKey, config) {
216
415
  return new GeminiVideoAdapter({ apiKey, ...config }, model);
@@ -1 +1 @@
1
- {"version":3,"file":"video.js","sources":["../../../src/adapters/video.ts"],"sourcesContent":["import {\n GenerateVideosOperation,\n VideoGenerationReferenceType,\n} from '@google/genai'\nimport { resolveMediaPrompt } from '@tanstack/ai'\nimport { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'\nimport { arrayBufferToBase64 } from '@tanstack/ai-utils'\nimport { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'\nimport { getGeminiVideoDurationOptions } from '../video/video-provider-options'\nimport type { DurationOptions } from '@tanstack/ai/adapters'\nimport type {\n ImagePart,\n MediaInputMetadata,\n VideoGenerationOptions,\n VideoJobResult,\n VideoStatusResult,\n VideoUrlResult,\n} from '@tanstack/ai'\nimport type {\n GenerateVideosConfig,\n GoogleGenAI,\n Image,\n VideoGenerationReferenceImage,\n} from '@google/genai'\nimport type {\n GeminiVideoModel,\n GeminiVideoModelDurationByName,\n GeminiVideoModelInputModalitiesByName,\n GeminiVideoModelProviderOptionsByName,\n GeminiVideoModelSizeByName,\n GeminiVideoProviderOptions,\n GeminiVideoSize,\n} from '../video/video-provider-options'\nimport type { GeminiClientConfig } from '../utils'\n\n/**\n * Configuration for Gemini video adapter.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GeminiVideoConfig extends GeminiClientConfig {}\n\n/**\n * Extract a human-readable message from a long-running operation's error,\n * which the SDK types as `Record<string, unknown>` (a google.rpc.Status).\n */\nfunction operationErrorMessage(error: Record<string, unknown>): string {\n if (typeof error.message === 'string' && error.message.length > 0) {\n return error.message\n }\n return JSON.stringify(error)\n}\n\n/**\n * Convert a TanStack image prompt part into the genai `Image` shape Veo\n * accepts: base64 `imageBytes` (data sources, data: URIs, fetched HTTP\n * URLs) or a `gcsUri` passthrough for Cloud Storage references.\n */\nasync function imagePartToVeoImage(\n part: ImagePart<MediaInputMetadata>,\n): Promise<Image> {\n if (part.source.type === 'data') {\n return {\n imageBytes: part.source.value,\n mimeType: part.source.mimeType || 'image/png',\n }\n }\n const url = part.source.value\n if (url.startsWith('gs://')) {\n return {\n gcsUri: url,\n ...(part.source.mimeType && { mimeType: part.source.mimeType }),\n }\n }\n if (url.startsWith('data:')) {\n const match = url.match(/^data:([^;,]+)?(;base64)?,(.*)$/)\n if (!match || !match[2]) {\n throw new Error(\n 'gemini: only base64 data: URIs are supported for video image inputs.',\n )\n }\n return {\n imageBytes: match[3] ?? '',\n mimeType: match[1] || part.source.mimeType || 'image/png',\n }\n }\n const response = await fetch(url)\n if (!response.ok) {\n throw new Error(\n `Failed to fetch image input (${response.status} ${response.statusText}): ${url}`,\n )\n }\n const blob = await response.blob()\n const buffer = await blob.arrayBuffer()\n return {\n imageBytes: arrayBufferToBase64(buffer),\n mimeType: part.source.mimeType || blob.type || 'image/png',\n }\n}\n\n/**\n * Gemini Veo Video Generation Adapter\n *\n * Tree-shakeable adapter for Google Veo video generation. Veo runs as a\n * long-running operation: `createVideoJob` starts the operation via the\n * `:predictLongRunning` endpoint, `getVideoStatus` polls it, and\n * `getVideoUrl` extracts the generated video's URI once it completes.\n *\n * Image prompt parts are routed by `metadata.role`:\n * - `'start_frame'` (or the first un-roled image) → the input image the\n * video starts from\n * - `'end_frame'` → `lastFrame` (the frame the video ends on)\n * - `'reference'` / `'character'` → `referenceImages` (asset references,\n * Veo 3.1)\n *\n * Note: the returned video URI is served by the Gemini Files API and\n * requires the API key (`x-goog-api-key` header or `?key=` query\n * parameter) to download.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport class GeminiVideoAdapter<\n TModel extends GeminiVideoModel,\n> extends BaseVideoAdapter<\n TModel,\n GeminiVideoProviderOptions,\n GeminiVideoModelProviderOptionsByName,\n GeminiVideoModelSizeByName,\n GeminiVideoModelInputModalitiesByName,\n GeminiVideoModelDurationByName\n> {\n readonly name = 'gemini' as const\n\n protected client: GoogleGenAI\n\n constructor(config: GeminiVideoConfig, model: TModel) {\n super({}, model)\n this.client = createGeminiClient(config)\n }\n\n async createVideoJob(\n options: VideoGenerationOptions<\n GeminiVideoProviderOptions,\n GeminiVideoSize,\n GeminiVideoModelDurationByName[TModel]\n >,\n ): Promise<VideoJobResult> {\n const { prompt, size, duration, modelOptions, logger } = options\n\n logger.request(\n `activity=video.create provider=${this.name} model=${this.model} size=${size ?? 'default'} duration=${duration ?? 'default'}`,\n { provider: this.name, model: this.model },\n )\n\n try {\n const resolved = resolveMediaPrompt(prompt)\n\n if (resolved.videos.length > 0) {\n throw new Error(\n `${this.name}.createVideoJob does not support video prompt parts (model: ${this.model}).`,\n )\n }\n if (resolved.audios.length > 0) {\n throw new Error(\n `${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`,\n )\n }\n\n const { image, lastFrame, referenceImages } = await this.routeImageParts(\n resolved.images,\n )\n\n const config: GenerateVideosConfig = {\n ...modelOptions,\n ...(size !== undefined && { aspectRatio: size }),\n ...(duration !== undefined && { durationSeconds: duration }),\n ...(lastFrame && { lastFrame }),\n ...(referenceImages.length > 0 && { referenceImages }),\n }\n\n const operation = await this.client.models.generateVideos({\n model: this.model,\n prompt: resolved.text,\n ...(image && { image }),\n config,\n })\n\n if (!operation.name) {\n throw new Error(\n 'Veo did not return an operation name for the video generation job.',\n )\n }\n\n return { jobId: operation.name, model: this.model }\n } catch (error) {\n logger.errors(`${this.name}.createVideoJob fatal`, {\n error,\n source: `${this.name}.createVideoJob`,\n })\n throw error\n }\n }\n\n /**\n * Route image prompt parts onto Veo's request fields by `metadata.role`.\n */\n private async routeImageParts(\n parts: Array<ImagePart<MediaInputMetadata>>,\n ): Promise<{\n image: Image | undefined\n lastFrame: Image | undefined\n referenceImages: Array<VideoGenerationReferenceImage>\n }> {\n let image: Image | undefined\n let lastFrame: Image | undefined\n const referenceImages: Array<VideoGenerationReferenceImage> = []\n\n for (const part of parts) {\n const role = part.metadata?.role\n switch (role) {\n case 'end_frame': {\n if (lastFrame) {\n throw new Error(\n `${this.name}: Veo accepts at most one 'end_frame' image.`,\n )\n }\n lastFrame = await imagePartToVeoImage(part)\n break\n }\n case 'reference':\n case 'character': {\n referenceImages.push({\n image: await imagePartToVeoImage(part),\n referenceType: VideoGenerationReferenceType.ASSET,\n })\n break\n }\n case 'start_frame':\n case undefined: {\n if (image) {\n throw new Error(\n `${this.name}: Veo accepts at most one starting image; received multiple 'start_frame'/un-roled images. Use metadata.role ('end_frame', 'reference') to disambiguate the others.`,\n )\n }\n image = await imagePartToVeoImage(part)\n break\n }\n case 'mask':\n case 'control':\n throw new Error(\n `${this.name}: unsupported image role \"${role}\" for Veo video generation.`,\n )\n }\n }\n\n return { image, lastFrame, referenceImages }\n }\n\n async getVideoStatus(jobId: string): Promise<VideoStatusResult> {\n const operation = await this.getOperation(jobId)\n\n if (!operation.done) {\n return { jobId, status: 'processing' }\n }\n\n if (operation.error) {\n return {\n jobId,\n status: 'failed',\n error: operationErrorMessage(operation.error),\n }\n }\n\n // The operation can finish \"successfully\" with every sample dropped by\n // Responsible-AI filters — surface that as a failure instead of letting\n // getVideoUrl() throw on an empty response.\n const videos = operation.response?.generatedVideos ?? []\n if (videos.length === 0) {\n const reasons = operation.response?.raiMediaFilteredReasons\n return {\n jobId,\n status: 'failed',\n error: reasons?.length\n ? `Video was filtered by Responsible-AI: ${reasons.join('; ')}`\n : 'Veo returned no generated videos.',\n }\n }\n\n return { jobId, status: 'completed' }\n }\n\n async getVideoUrl(jobId: string): Promise<VideoUrlResult> {\n const operation = await this.getOperation(jobId)\n\n if (!operation.done) {\n throw new Error(\n `Video is not ready yet. Check status first. Job ID: ${jobId}`,\n )\n }\n\n if (operation.error) {\n throw new Error(\n `Video generation failed: ${operationErrorMessage(operation.error)}`,\n )\n }\n\n const uri = operation.response?.generatedVideos?.[0]?.video?.uri\n if (!uri) {\n const reasons = operation.response?.raiMediaFilteredReasons\n throw new Error(\n reasons?.length\n ? `Video was filtered by Responsible-AI: ${reasons.join('; ')}`\n : `Video URL not found in operation response. Job ID: ${jobId}`,\n )\n }\n\n return { jobId, url: uri }\n }\n\n override availableDurations(): DurationOptions<\n GeminiVideoModelDurationByName[TModel]\n > {\n return getGeminiVideoDurationOptions(this.model)\n }\n\n override snapDuration(\n seconds: number,\n ): GeminiVideoModelDurationByName[TModel] | undefined {\n return snapToDurationOption(seconds, this.availableDurations())\n }\n\n /**\n * Fetch the long-running operation by name. The SDK's\n * `operations.getVideosOperation` needs a real `GenerateVideosOperation`\n * instance (it calls `_fromAPIResponse` on it), so reconstruct one from\n * the job ID rather than passing an object literal.\n */\n private async getOperation(jobId: string): Promise<GenerateVideosOperation> {\n const operation = new GenerateVideosOperation()\n operation.name = jobId\n return await this.client.operations.getVideosOperation({ operation })\n }\n}\n\n/**\n * Creates a Gemini video adapter with an explicit API key.\n * Type resolution happens here at the call site.\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'veo-3.1-generate-preview')\n * @param apiKey - Your Google API key\n * @param config - Optional additional configuration\n * @returns Configured Gemini video adapter instance with resolved types\n *\n * @example\n * ```typescript\n * const adapter = createGeminiVideo('veo-3.1-generate-preview', 'your-api-key');\n *\n * const { jobId } = await generateVideo({\n * adapter,\n * prompt: 'A beautiful sunset over the ocean',\n * duration: adapter.snapDuration(7), // → 6\n * });\n * ```\n */\nexport function createGeminiVideo<TModel extends GeminiVideoModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GeminiVideoConfig, 'apiKey'>,\n): GeminiVideoAdapter<TModel> {\n return new GeminiVideoAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Gemini video adapter with automatic API key detection from environment variables.\n * Type resolution happens here at the call site.\n *\n * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (Browser with injected env)\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'veo-3.1-generate-preview')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Gemini video adapter instance with resolved types\n * @throws Error if GOOGLE_API_KEY or GEMINI_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * // Automatically uses GOOGLE_API_KEY from environment\n * const adapter = geminiVideo('veo-3.1-generate-preview');\n *\n * // Create a video generation job\n * const { jobId } = await generateVideo({\n * adapter,\n * prompt: 'A cat playing piano'\n * });\n *\n * // Poll for status\n * const status = await getVideoJobStatus({ adapter, jobId });\n * ```\n */\nexport function geminiVideo<TModel extends GeminiVideoModel>(\n model: TModel,\n config?: Omit<GeminiVideoConfig, 'apiKey'>,\n): GeminiVideoAdapter<TModel> {\n const apiKey = getGeminiApiKeyFromEnv()\n return createGeminiVideo(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;;;;AA8CA,SAAS,sBAAsB,OAAwC;AACrE,MAAI,OAAO,MAAM,YAAY,YAAY,MAAM,QAAQ,SAAS,GAAG;AACjE,WAAO,MAAM;AAAA,EACf;AACA,SAAO,KAAK,UAAU,KAAK;AAC7B;AAOA,eAAe,oBACb,MACgB;AAChB,MAAI,KAAK,OAAO,SAAS,QAAQ;AAC/B,WAAO;AAAA,MACL,YAAY,KAAK,OAAO;AAAA,MACxB,UAAU,KAAK,OAAO,YAAY;AAAA,IAAA;AAAA,EAEtC;AACA,QAAM,MAAM,KAAK,OAAO;AACxB,MAAI,IAAI,WAAW,OAAO,GAAG;AAC3B,WAAO;AAAA,MACL,QAAQ;AAAA,MACR,GAAI,KAAK,OAAO,YAAY,EAAE,UAAU,KAAK,OAAO,SAAA;AAAA,IAAS;AAAA,EAEjE;AACA,MAAI,IAAI,WAAW,OAAO,GAAG;AAC3B,UAAM,QAAQ,IAAI,MAAM,iCAAiC;AACzD,QAAI,CAAC,SAAS,CAAC,MAAM,CAAC,GAAG;AACvB,YAAM,IAAI;AAAA,QACR;AAAA,MAAA;AAAA,IAEJ;AACA,WAAO;AAAA,MACL,YAAY,MAAM,CAAC,KAAK;AAAA,MACxB,UAAU,MAAM,CAAC,KAAK,KAAK,OAAO,YAAY;AAAA,IAAA;AAAA,EAElD;AACA,QAAM,WAAW,MAAM,MAAM,GAAG;AAChC,MAAI,CAAC,SAAS,IAAI;AAChB,UAAM,IAAI;AAAA,MACR,gCAAgC,SAAS,MAAM,IAAI,SAAS,UAAU,MAAM,GAAG;AAAA,IAAA;AAAA,EAEnF;AACA,QAAM,OAAO,MAAM,SAAS,KAAA;AAC5B,QAAM,SAAS,MAAM,KAAK,YAAA;AAC1B,SAAO;AAAA,IACL,YAAY,oBAAoB,MAAM;AAAA,IACtC,UAAU,KAAK,OAAO,YAAY,KAAK,QAAQ;AAAA,EAAA;AAEnD;AAuBO,MAAM,2BAEH,iBAOR;AAAA,EACS,OAAO;AAAA,EAEN;AAAA,EAEV,YAAY,QAA2B,OAAe;AACpD,UAAM,CAAA,GAAI,KAAK;AACf,SAAK,SAAS,mBAAmB,MAAM;AAAA,EACzC;AAAA,EAEA,MAAM,eACJ,SAKyB;AACzB,UAAM,EAAE,QAAQ,MAAM,UAAU,cAAc,WAAW;AAEzD,WAAO;AAAA,MACL,kCAAkC,KAAK,IAAI,UAAU,KAAK,KAAK,SAAS,QAAQ,SAAS,aAAa,YAAY,SAAS;AAAA,MAC3H,EAAE,UAAU,KAAK,MAAM,OAAO,KAAK,MAAA;AAAA,IAAM;AAG3C,QAAI;AACF,YAAM,WAAW,mBAAmB,MAAM;AAE1C,UAAI,SAAS,OAAO,SAAS,GAAG;AAC9B,cAAM,IAAI;AAAA,UACR,GAAG,KAAK,IAAI,+DAA+D,KAAK,KAAK;AAAA,QAAA;AAAA,MAEzF;AACA,UAAI,SAAS,OAAO,SAAS,GAAG;AAC9B,cAAM,IAAI;AAAA,UACR,GAAG,KAAK,IAAI,+DAA+D,KAAK,KAAK;AAAA,QAAA;AAAA,MAEzF;AAEA,YAAM,EAAE,OAAO,WAAW,gBAAA,IAAoB,MAAM,KAAK;AAAA,QACvD,SAAS;AAAA,MAAA;AAGX,YAAM,SAA+B;AAAA,QACnC,GAAG;AAAA,QACH,GAAI,SAAS,UAAa,EAAE,aAAa,KAAA;AAAA,QACzC,GAAI,aAAa,UAAa,EAAE,iBAAiB,SAAA;AAAA,QACjD,GAAI,aAAa,EAAE,UAAA;AAAA,QACnB,GAAI,gBAAgB,SAAS,KAAK,EAAE,gBAAA;AAAA,MAAgB;AAGtD,YAAM,YAAY,MAAM,KAAK,OAAO,OAAO,eAAe;AAAA,QACxD,OAAO,KAAK;AAAA,QACZ,QAAQ,SAAS;AAAA,QACjB,GAAI,SAAS,EAAE,MAAA;AAAA,QACf;AAAA,MAAA,CACD;AAED,UAAI,CAAC,UAAU,MAAM;AACnB,cAAM,IAAI;AAAA,UACR;AAAA,QAAA;AAAA,MAEJ;AAEA,aAAO,EAAE,OAAO,UAAU,MAAM,OAAO,KAAK,MAAA;AAAA,IAC9C,SAAS,OAAO;AACd,aAAO,OAAO,GAAG,KAAK,IAAI,yBAAyB;AAAA,QACjD;AAAA,QACA,QAAQ,GAAG,KAAK,IAAI;AAAA,MAAA,CACrB;AACD,YAAM;AAAA,IACR;AAAA,EACF;AAAA;AAAA;AAAA;AAAA,EAKA,MAAc,gBACZ,OAKC;AACD,QAAI;AACJ,QAAI;AACJ,UAAM,kBAAwD,CAAA;AAE9D,eAAW,QAAQ,OAAO;AACxB,YAAM,OAAO,KAAK,UAAU;AAC5B,cAAQ,MAAA;AAAA,QACN,KAAK,aAAa;AAChB,cAAI,WAAW;AACb,kBAAM,IAAI;AAAA,cACR,GAAG,KAAK,IAAI;AAAA,YAAA;AAAA,UAEhB;AACA,sBAAY,MAAM,oBAAoB,IAAI;AAC1C;AAAA,QACF;AAAA,QACA,KAAK;AAAA,QACL,KAAK,aAAa;AAChB,0BAAgB,KAAK;AAAA,YACnB,OAAO,MAAM,oBAAoB,IAAI;AAAA,YACrC,eAAe,6BAA6B;AAAA,UAAA,CAC7C;AACD;AAAA,QACF;AAAA,QACA,KAAK;AAAA,QACL,KAAK,QAAW;AACd,cAAI,OAAO;AACT,kBAAM,IAAI;AAAA,cACR,GAAG,KAAK,IAAI;AAAA,YAAA;AAAA,UAEhB;AACA,kBAAQ,MAAM,oBAAoB,IAAI;AACtC;AAAA,QACF;AAAA,QACA,KAAK;AAAA,QACL,KAAK;AACH,gBAAM,IAAI;AAAA,YACR,GAAG,KAAK,IAAI,6BAA6B,IAAI;AAAA,UAAA;AAAA,MAC/C;AAAA,IAEN;AAEA,WAAO,EAAE,OAAO,WAAW,gBAAA;AAAA,EAC7B;AAAA,EAEA,MAAM,eAAe,OAA2C;AAC9D,UAAM,YAAY,MAAM,KAAK,aAAa,KAAK;AAE/C,QAAI,CAAC,UAAU,MAAM;AACnB,aAAO,EAAE,OAAO,QAAQ,aAAA;AAAA,IAC1B;AAEA,QAAI,UAAU,OAAO;AACnB,aAAO;AAAA,QACL;AAAA,QACA,QAAQ;AAAA,QACR,OAAO,sBAAsB,UAAU,KAAK;AAAA,MAAA;AAAA,IAEhD;AAKA,UAAM,SAAS,UAAU,UAAU,mBAAmB,CAAA;AACtD,QAAI,OAAO,WAAW,GAAG;AACvB,YAAM,UAAU,UAAU,UAAU;AACpC,aAAO;AAAA,QACL;AAAA,QACA,QAAQ;AAAA,QACR,OAAO,SAAS,SACZ,yCAAyC,QAAQ,KAAK,IAAI,CAAC,KAC3D;AAAA,MAAA;AAAA,IAER;AAEA,WAAO,EAAE,OAAO,QAAQ,YAAA;AAAA,EAC1B;AAAA,EAEA,MAAM,YAAY,OAAwC;AACxD,UAAM,YAAY,MAAM,KAAK,aAAa,KAAK;AAE/C,QAAI,CAAC,UAAU,MAAM;AACnB,YAAM,IAAI;AAAA,QACR,uDAAuD,KAAK;AAAA,MAAA;AAAA,IAEhE;AAEA,QAAI,UAAU,OAAO;AACnB,YAAM,IAAI;AAAA,QACR,4BAA4B,sBAAsB,UAAU,KAAK,CAAC;AAAA,MAAA;AAAA,IAEtE;AAEA,UAAM,MAAM,UAAU,UAAU,kBAAkB,CAAC,GAAG,OAAO;AAC7D,QAAI,CAAC,KAAK;AACR,YAAM,UAAU,UAAU,UAAU;AACpC,YAAM,IAAI;AAAA,QACR,SAAS,SACL,yCAAyC,QAAQ,KAAK,IAAI,CAAC,KAC3D,sDAAsD,KAAK;AAAA,MAAA;AAAA,IAEnE;AAEA,WAAO,EAAE,OAAO,KAAK,IAAA;AAAA,EACvB;AAAA,EAES,qBAEP;AACA,WAAO,8BAA8B,KAAK,KAAK;AAAA,EACjD;AAAA,EAES,aACP,SACoD;AACpD,WAAO,qBAAqB,SAAS,KAAK,mBAAA,CAAoB;AAAA,EAChE;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAQA,MAAc,aAAa,OAAiD;AAC1E,UAAM,YAAY,IAAI,wBAAA;AACtB,cAAU,OAAO;AACjB,WAAO,MAAM,KAAK,OAAO,WAAW,mBAAmB,EAAE,WAAW;AAAA,EACtE;AACF;AAwBO,SAAS,kBACd,OACA,QACA,QAC4B;AAC5B,SAAO,IAAI,mBAAmB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAC5D;AAgCO,SAAS,YACd,OACA,QAC4B;AAC5B,QAAM,SAAS,uBAAA;AACf,SAAO,kBAAkB,OAAO,QAAQ,MAAM;AAChD;"}
1
+ {"version":3,"file":"video.js","sources":["../../../src/adapters/video.ts"],"sourcesContent":["import {\n GenerateVideosOperation,\n VideoGenerationReferenceType,\n} from '@google/genai'\nimport { resolveMediaPrompt } from '@tanstack/ai'\nimport { BaseVideoAdapter, snapToDurationOption } from '@tanstack/ai/adapters'\nimport { arrayBufferToBase64 } from '@tanstack/ai-utils'\nimport { createGeminiClient, getGeminiApiKeyFromEnv } from '../utils'\nimport {\n getGeminiVideoDurationOptions,\n isInteractionsVideoModel,\n} from '../video/video-provider-options'\nimport type { DurationOptions } from '@tanstack/ai/adapters'\nimport type {\n ImagePart,\n MediaInputMetadata,\n TokenUsage,\n VideoGenerationOptions,\n VideoJobResult,\n VideoPart,\n VideoStatusResult,\n VideoUrlResult,\n} from '@tanstack/ai'\nimport type {\n GenerateVideosConfig,\n GoogleGenAI,\n Image,\n Interactions,\n VideoGenerationReferenceImage,\n} from '@google/genai'\nimport type {\n GeminiOmniVideoProviderOptions,\n GeminiVideoModel,\n GeminiVideoModelDurationByName,\n GeminiVideoModelInputModalitiesByName,\n GeminiVideoModelProviderOptionsByName,\n GeminiVideoModelSizeByName,\n GeminiVideoProviderOptions,\n GeminiVideoSize,\n} from '../video/video-provider-options'\nimport type { GeminiClientConfig } from '../utils/client'\n\ntype Interaction = Interactions.Interaction\ntype InteractionContent = Interactions.Content\n\n/**\n * Configuration for Gemini video adapter.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport interface GeminiVideoConfig extends GeminiClientConfig {\n /**\n * Opt into fetching HTTP(S) image URL inputs. Veo's predict API accepts\n * only inline `imageBytes` or a `gcsUri`, so an HTTP(S) URL has to be\n * downloaded and base64-encoded locally — which buffers the whole image in\n * memory and can OOM constrained runtimes (e.g. Cloudflare Workers). When\n * `false` (the default), HTTP(S) URL image inputs throw; pass a `data:` URI\n * or a `gs://` reference, or set this to `true` to opt into buffering.\n */\n allowUrlFetch?: boolean\n}\n\n/**\n * Extract a human-readable message from a long-running operation's error,\n * which the SDK types as `Record<string, unknown>` (a google.rpc.Status).\n */\nfunction operationErrorMessage(error: Record<string, unknown>): string {\n if (typeof error.message === 'string' && error.message.length > 0) {\n return error.message\n }\n return JSON.stringify(error)\n}\n\n/**\n * Convert a TanStack image prompt part into the genai `Image` shape Veo\n * accepts: base64 `imageBytes` (data sources, data: URIs, fetched HTTP\n * URLs) or a `gcsUri` passthrough for Cloud Storage references.\n *\n * Unlike `generateContent` (chat / native image generation), Veo's predict\n * API has no `fileData.fileUri` equivalent — `Image` only accepts\n * `imageBytes` or `gcsUri`. An HTTP(S) URL therefore has to be fetched and\n * inlined locally, which buffers the whole image in memory; that only happens\n * when the caller opts in via `allowUrlFetch`, otherwise it throws. Prefer a\n * `gs://` reference on memory-constrained runtimes.\n */\nasync function imagePartToVeoImage(\n part: ImagePart<MediaInputMetadata>,\n allowUrlFetch: boolean,\n): Promise<Image> {\n if (part.source.type === 'data') {\n return {\n imageBytes: part.source.value,\n mimeType: part.source.mimeType || 'image/png',\n }\n }\n const url = part.source.value\n if (url.startsWith('gs://')) {\n return {\n gcsUri: url,\n ...(part.source.mimeType && { mimeType: part.source.mimeType }),\n }\n }\n if (url.startsWith('data:')) {\n const match = url.match(/^data:([^;,]+)?(;base64)?,(.*)$/)\n if (!match || !match[2]) {\n throw new Error(\n 'gemini: only base64 data: URIs are supported for video image inputs.',\n )\n }\n return {\n imageBytes: match[3] ?? '',\n mimeType: match[1] || part.source.mimeType || 'image/png',\n }\n }\n if (!allowUrlFetch) {\n throw new Error(\n `gemini Veo: HTTP(S) URL image inputs are not fetched by default because ` +\n `Veo accepts only inline bytes, so the image would be downloaded and ` +\n `buffered in memory (risking OOM on constrained runtimes). Pass a ` +\n `data: URI or a gs:// reference, or set \\`allowUrlFetch: true\\` on the ` +\n `adapter config to opt into fetching. URL: ${url}`,\n )\n }\n const response = await fetch(url)\n if (!response.ok) {\n throw new Error(\n `Failed to fetch image input (${response.status} ${response.statusText}): ${url}`,\n )\n }\n const blob = await response.blob()\n const buffer = await blob.arrayBuffer()\n return {\n imageBytes: arrayBufferToBase64(buffer),\n mimeType: part.source.mimeType || blob.type || 'image/png',\n }\n}\n\n/**\n * Convert an image or video prompt part into an Interactions API content\n * block. Data sources become inline base64 `data`; URL sources pass through\n * as `uri` (Files API URIs — mirrors the Interactions text adapter).\n */\nfunction mediaPartToInteractionsContent(\n part: ImagePart<MediaInputMetadata> | VideoPart<MediaInputMetadata>,\n): InteractionContent {\n const mimeType = part.source.mimeType\n if (part.type === 'image') {\n return part.source.type === 'data'\n ? { type: 'image', data: part.source.value, mime_type: mimeType }\n : { type: 'image', uri: part.source.value, mime_type: mimeType }\n }\n return part.source.type === 'data'\n ? { type: 'video', data: part.source.value, mime_type: mimeType }\n : { type: 'video', uri: part.source.value, mime_type: mimeType }\n}\n\n/**\n * Pull the generated video out of a completed interaction. Prefers the\n * SDK's `output_video` sugar, then walks `steps` back-to-front for the last\n * `model_output` step carrying a video content block (the wire shape the\n * raw REST response uses).\n */\nfunction extractInteractionVideo(\n interaction: Interaction,\n): { data?: string; uri?: string; mimeType: string } | undefined {\n const direct = interaction.output_video\n if (direct && (direct.data || direct.uri)) {\n return {\n data: direct.data,\n uri: direct.uri,\n mimeType: direct.mime_type || 'video/mp4',\n }\n }\n const steps = interaction.steps ?? []\n for (let i = steps.length - 1; i >= 0; i--) {\n const step = steps[i]\n if (step?.type !== 'model_output') continue\n for (const block of step.content ?? []) {\n if (block.type === 'video' && (block.data || block.uri)) {\n return {\n data: block.data,\n uri: block.uri,\n mimeType: block.mime_type || 'video/mp4',\n }\n }\n }\n }\n return undefined\n}\n\n/**\n * Map Interactions usage onto the canonical TokenUsage shape. Omni reports\n * video output via `output_tokens_by_modality`; fall back to the video\n * modality entry when the total is absent.\n */\nfunction interactionUsageToTokenUsage(\n usage: Interaction['usage'],\n): TokenUsage | undefined {\n if (!usage) return undefined\n const videoTokens = usage.output_tokens_by_modality?.find(\n (entry) => entry.modality === 'video',\n )?.tokens\n const promptTokens = usage.total_input_tokens ?? 0\n const completionTokens = usage.total_output_tokens ?? videoTokens ?? 0\n return {\n promptTokens,\n completionTokens,\n totalTokens: usage.total_tokens ?? promptTokens + completionTokens,\n }\n}\n\n/**\n * Gemini Video Generation Adapter (Veo + Gemini Omni Flash)\n *\n * Tree-shakeable adapter for Google video generation, routing by model:\n *\n * **Veo models** run as a long-running operation: `createVideoJob` starts\n * the operation via the `:predictLongRunning` endpoint, `getVideoStatus`\n * polls it, and `getVideoUrl` extracts the generated video's URI once it\n * completes. Image prompt parts are routed by `metadata.role`:\n * - `'start_frame'` (or the first un-roled image) → the input image the\n * video starts from\n * - `'end_frame'` → `lastFrame` (the frame the video ends on)\n * - `'reference'` / `'character'` → `referenceImages` (asset references,\n * Veo 3.1)\n *\n * Note: the returned Veo video URI is served by the Gemini Files API and\n * requires the API key (`x-goog-api-key` header or `?key=` query\n * parameter) to download.\n *\n * **Gemini Omni Flash** (`gemini-omni-flash-preview`) only serves the\n * Interactions API: `createVideoJob` creates a background interaction with\n * `response_modalities: ['video']`, `getVideoStatus` polls it by id, and\n * `getVideoUrl` returns the inline base64 MP4 as a `data:` URL (or the\n * Files API URI when the server delivers by reference). Image and video\n * prompt parts are sent as interaction content blocks, grouped as images,\n * then videos, then the text prompt (interleaving is not preserved); pass\n * `modelOptions.previous_interaction_id` to conversationally edit a prior\n * Omni generation.\n *\n * @experimental Video generation is an experimental feature and may change.\n */\nexport class GeminiVideoAdapter<\n TModel extends GeminiVideoModel,\n> extends BaseVideoAdapter<\n TModel,\n GeminiVideoModelProviderOptionsByName[TModel],\n GeminiVideoModelProviderOptionsByName,\n GeminiVideoModelSizeByName,\n GeminiVideoModelInputModalitiesByName,\n GeminiVideoModelDurationByName\n> {\n readonly name = 'gemini' as const\n\n protected client: GoogleGenAI\n private readonly allowUrlFetch: boolean\n\n constructor(config: GeminiVideoConfig, model: TModel) {\n super({}, model)\n this.client = createGeminiClient(config)\n this.allowUrlFetch = config.allowUrlFetch ?? false\n }\n\n async createVideoJob(\n options: VideoGenerationOptions<\n GeminiVideoModelProviderOptionsByName[TModel],\n GeminiVideoSize,\n GeminiVideoModelDurationByName[TModel]\n >,\n ): Promise<VideoJobResult> {\n const { prompt, size, duration, logger } = options\n\n logger.request(\n `activity=video.create provider=${this.name} model=${this.model} size=${size ?? 'default'} duration=${duration ?? 'default'}`,\n { provider: this.name, model: this.model },\n )\n\n if (isInteractionsVideoModel(this.model)) {\n return await this.createInteractionsVideoJob(options)\n }\n const modelOptions = options.modelOptions as\n | GeminiVideoProviderOptions\n | undefined\n\n try {\n const resolved = resolveMediaPrompt(prompt)\n\n if (resolved.videos.length > 0) {\n throw new Error(\n `${this.name}.createVideoJob does not support video prompt parts (model: ${this.model}).`,\n )\n }\n if (resolved.audios.length > 0) {\n throw new Error(\n `${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`,\n )\n }\n\n const { image, lastFrame, referenceImages } = await this.routeImageParts(\n resolved.images,\n )\n\n const config: GenerateVideosConfig = {\n ...modelOptions,\n ...(size !== undefined && { aspectRatio: size }),\n ...(duration !== undefined && { durationSeconds: duration }),\n ...(lastFrame && { lastFrame }),\n ...(referenceImages.length > 0 && { referenceImages }),\n }\n\n const operation = await this.client.models.generateVideos({\n model: this.model,\n prompt: resolved.text,\n ...(image && { image }),\n config,\n })\n\n if (!operation.name) {\n throw new Error(\n 'Veo did not return an operation name for the video generation job.',\n )\n }\n\n return { jobId: operation.name, model: this.model }\n } catch (error) {\n logger.errors(`${this.name}.createVideoJob fatal`, {\n error,\n source: `${this.name}.createVideoJob`,\n })\n throw error\n }\n }\n\n /**\n * Gemini Omni Flash job creation via the Interactions API. Creates a\n * background interaction requesting video output; the interaction id is\n * the job id polled by `getVideoStatus` / `getVideoUrl`.\n */\n private async createInteractionsVideoJob(\n options: VideoGenerationOptions<\n GeminiVideoModelProviderOptionsByName[TModel],\n GeminiVideoSize,\n GeminiVideoModelDurationByName[TModel]\n >,\n ): Promise<VideoJobResult> {\n const { prompt, size, duration, logger } = options\n const modelOptions = options.modelOptions as\n | GeminiOmniVideoProviderOptions\n | undefined\n\n try {\n const resolved = resolveMediaPrompt(prompt)\n\n if (resolved.audios.length > 0) {\n throw new Error(\n `${this.name}.createVideoJob does not support audio prompt parts (model: ${this.model}).`,\n )\n }\n\n const content: Array<InteractionContent> = [\n ...resolved.images.map(mediaPartToInteractionsContent),\n ...resolved.videos.map(mediaPartToInteractionsContent),\n ]\n if (resolved.text) {\n content.push({ type: 'text', text: resolved.text })\n }\n if (content.length === 0) {\n throw new Error(\n `${this.name}.createVideoJob: the prompt produced no content to send (model: ${this.model}).`,\n )\n }\n\n // Reject out-of-range durations locally rather than snapping (which\n // would silently change the clip length the caller asked for) or\n // letting the live API reject them after the round trip.\n const durations = this.availableDurations()\n if (\n duration !== undefined &&\n durations.kind === 'range' &&\n (duration < durations.min || duration > durations.max)\n ) {\n throw new Error(\n `${this.name}.createVideoJob: duration ${duration}s is outside the ${durations.min}–${durations.max}s range supported by ${this.model}. Use snapDuration() to snap arbitrary values into range.`,\n )\n }\n\n // Aspect ratio and clip length ride on `response_format`. Duration is\n // a `\"<seconds>s\"` string, accepted anywhere in the 3–10s range\n // (fractional included) and defaulting to 10s when omitted — verified\n // against the live API; the docs don't publish the range constraints.\n const responseFormat =\n size !== undefined || duration !== undefined\n ? {\n response_format: {\n type: 'video' as const,\n ...(size !== undefined && { aspect_ratio: size }),\n ...(duration !== undefined && { duration: `${duration}s` }),\n },\n }\n : {}\n\n const interaction = await this.client.interactions.create({\n ...modelOptions,\n model: this.model,\n input: [{ type: 'user_input', content }],\n response_modalities: ['video'],\n background: true,\n ...responseFormat,\n })\n\n if (!interaction.id) {\n throw new Error(\n 'Gemini Omni did not return an interaction id for the video generation job.',\n )\n }\n\n return { jobId: interaction.id, model: this.model }\n } catch (error) {\n logger.errors(`${this.name}.createVideoJob fatal`, {\n error,\n source: `${this.name}.createVideoJob`,\n })\n throw error\n }\n }\n\n /**\n * Route image prompt parts onto Veo's request fields by `metadata.role`.\n */\n private async routeImageParts(\n parts: Array<ImagePart<MediaInputMetadata>>,\n ): Promise<{\n image: Image | undefined\n lastFrame: Image | undefined\n referenceImages: Array<VideoGenerationReferenceImage>\n }> {\n let image: Image | undefined\n let lastFrame: Image | undefined\n const referenceImages: Array<VideoGenerationReferenceImage> = []\n\n for (const part of parts) {\n const role = part.metadata?.role\n switch (role) {\n case 'end_frame': {\n if (lastFrame) {\n throw new Error(\n `${this.name}: Veo accepts at most one 'end_frame' image.`,\n )\n }\n lastFrame = await imagePartToVeoImage(part, this.allowUrlFetch)\n break\n }\n case 'reference':\n case 'character': {\n referenceImages.push({\n image: await imagePartToVeoImage(part, this.allowUrlFetch),\n referenceType: VideoGenerationReferenceType.ASSET,\n })\n break\n }\n case 'start_frame':\n case undefined: {\n if (image) {\n throw new Error(\n `${this.name}: Veo accepts at most one starting image; received multiple 'start_frame'/un-roled images. Use metadata.role ('end_frame', 'reference') to disambiguate the others.`,\n )\n }\n image = await imagePartToVeoImage(part, this.allowUrlFetch)\n break\n }\n case 'mask':\n case 'control':\n throw new Error(\n `${this.name}: unsupported image role \"${role}\" for Veo video generation.`,\n )\n }\n }\n\n return { image, lastFrame, referenceImages }\n }\n\n async getVideoStatus(jobId: string): Promise<VideoStatusResult> {\n if (isInteractionsVideoModel(this.model)) {\n return await this.getInteractionsVideoStatus(jobId)\n }\n const operation = await this.getOperation(jobId)\n\n if (!operation.done) {\n return { jobId, status: 'processing' }\n }\n\n if (operation.error) {\n return {\n jobId,\n status: 'failed',\n error: operationErrorMessage(operation.error),\n }\n }\n\n // The operation can finish \"successfully\" with every sample dropped by\n // Responsible-AI filters — surface that as a failure instead of letting\n // getVideoUrl() throw on an empty response.\n const videos = operation.response?.generatedVideos ?? []\n if (videos.length === 0) {\n const reasons = operation.response?.raiMediaFilteredReasons\n return {\n jobId,\n status: 'failed',\n error: reasons?.length\n ? `Video was filtered by Responsible-AI: ${reasons.join('; ')}`\n : 'Veo returned no generated videos.',\n }\n }\n\n return { jobId, status: 'completed' }\n }\n\n /**\n * Poll an Omni background interaction. `in_progress` maps to\n * 'processing'; a `completed` interaction with no video content (e.g.\n * filtered output) is surfaced as a failure so `getVideoUrl` doesn't\n * throw on an empty response. `requires_action` also fails: the adapter\n * never sends tools, so it can only arise via\n * `previous_interaction_id` chaining onto a tool-bearing interaction —\n * and such an interaction never progresses without a client response,\n * so polling it would spin until timeout.\n */\n private async getInteractionsVideoStatus(\n jobId: string,\n ): Promise<VideoStatusResult> {\n const interaction = await this.getInteraction(jobId)\n const status = interaction.status\n\n if (status === 'in_progress') {\n return { jobId, status: 'processing' }\n }\n if (status === 'requires_action') {\n return {\n jobId,\n status: 'failed',\n error:\n 'Gemini Omni interaction is waiting on a client action (tool response), which the video jobs flow does not support.',\n }\n }\n if (status === 'completed') {\n if (!extractInteractionVideo(interaction)) {\n return {\n jobId,\n status: 'failed',\n error:\n 'Gemini Omni completed the interaction without returning a video (the output may have been filtered).',\n }\n }\n return { jobId, status: 'completed' }\n }\n return {\n jobId,\n status: 'failed',\n error: `Gemini Omni video generation ended with status \"${status}\".`,\n }\n }\n\n async getVideoUrl(jobId: string): Promise<VideoUrlResult> {\n if (isInteractionsVideoModel(this.model)) {\n return await this.getInteractionsVideoUrl(jobId)\n }\n const operation = await this.getOperation(jobId)\n\n if (!operation.done) {\n throw new Error(\n `Video is not ready yet. Check status first. Job ID: ${jobId}`,\n )\n }\n\n if (operation.error) {\n throw new Error(\n `Video generation failed: ${operationErrorMessage(operation.error)}`,\n )\n }\n\n const uri = operation.response?.generatedVideos?.[0]?.video?.uri\n if (!uri) {\n const reasons = operation.response?.raiMediaFilteredReasons\n throw new Error(\n reasons?.length\n ? `Video was filtered by Responsible-AI: ${reasons.join('; ')}`\n : `Video URL not found in operation response. Job ID: ${jobId}`,\n )\n }\n\n return { jobId, url: uri }\n }\n\n /**\n * Extract the finished Omni video. Inline base64 output (the API default)\n * becomes a `data:` URL — matching the OpenAI Sora adapter's inline\n * delivery — and URI delivery passes through (Files API URIs need the API\n * key to download, like Veo). Usage carries the video-modality output\n * tokens (Omni bills per second of video, reported as tokens).\n */\n private async getInteractionsVideoUrl(\n jobId: string,\n ): Promise<VideoUrlResult> {\n const interaction = await this.getInteraction(jobId)\n const status = interaction.status\n\n if (status === 'in_progress') {\n throw new Error(\n `Video is not ready yet. Check status first. Job ID: ${jobId}`,\n )\n }\n if (status !== 'completed') {\n throw new Error(\n `Video generation failed: Gemini Omni interaction ended with status \"${status}\". Job ID: ${jobId}`,\n )\n }\n\n const video = extractInteractionVideo(interaction)\n if (!video) {\n throw new Error(\n `Video not found in interaction response (the output may have been filtered). Job ID: ${jobId}`,\n )\n }\n\n const usage = interactionUsageToTokenUsage(interaction.usage)\n const url = video.uri ?? `data:${video.mimeType};base64,${video.data}`\n return { jobId, url, ...(usage && { usage }) }\n }\n\n override availableDurations(): DurationOptions<\n GeminiVideoModelDurationByName[TModel]\n > {\n return getGeminiVideoDurationOptions(this.model)\n }\n\n override snapDuration(\n seconds: number,\n ): GeminiVideoModelDurationByName[TModel] | undefined {\n return snapToDurationOption(seconds, this.availableDurations())\n }\n\n /**\n * Fetch the long-running operation by name. The SDK's\n * `operations.getVideosOperation` needs a real `GenerateVideosOperation`\n * instance (it calls `_fromAPIResponse` on it), so reconstruct one from\n * the job ID rather than passing an object literal.\n */\n private async getOperation(jobId: string): Promise<GenerateVideosOperation> {\n const operation = new GenerateVideosOperation()\n operation.name = jobId\n return await this.client.operations.getVideosOperation({ operation })\n }\n\n /**\n * Fetch an Omni background interaction by id.\n */\n private async getInteraction(jobId: string): Promise<Interaction> {\n return await this.client.interactions.get(jobId)\n }\n}\n\n/**\n * Creates a Gemini video adapter with an explicit API key.\n * Type resolution happens here at the call site.\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'veo-3.1-generate-preview')\n * @param apiKey - Your Google API key\n * @param config - Optional additional configuration\n * @returns Configured Gemini video adapter instance with resolved types\n *\n * @example\n * ```typescript\n * const adapter = createGeminiVideo('veo-3.1-generate-preview', 'your-api-key');\n *\n * const { jobId } = await generateVideo({\n * adapter,\n * prompt: 'A beautiful sunset over the ocean',\n * duration: adapter.snapDuration(7), // → 6\n * });\n * ```\n */\nexport function createGeminiVideo<TModel extends GeminiVideoModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GeminiVideoConfig, 'apiKey'>,\n): GeminiVideoAdapter<TModel> {\n return new GeminiVideoAdapter({ apiKey, ...config }, model)\n}\n\n/**\n * Creates a Gemini video adapter with automatic API key detection from environment variables.\n * Type resolution happens here at the call site.\n *\n * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:\n * - `process.env` (Node.js)\n * - `window.env` (Browser with injected env)\n *\n * @experimental Video generation is an experimental feature and may change.\n *\n * @param model - The model name (e.g., 'veo-3.1-generate-preview')\n * @param config - Optional configuration (excluding apiKey which is auto-detected)\n * @returns Configured Gemini video adapter instance with resolved types\n * @throws Error if GOOGLE_API_KEY or GEMINI_API_KEY is not found in environment\n *\n * @example\n * ```typescript\n * // Automatically uses GOOGLE_API_KEY from environment\n * const adapter = geminiVideo('veo-3.1-generate-preview');\n *\n * // Create a video generation job\n * const { jobId } = await generateVideo({\n * adapter,\n * prompt: 'A cat playing piano'\n * });\n *\n * // Poll for status\n * const status = await getVideoJobStatus({ adapter, jobId });\n * ```\n */\nexport function geminiVideo<TModel extends GeminiVideoModel>(\n model: TModel,\n config?: Omit<GeminiVideoConfig, 'apiKey'>,\n): GeminiVideoAdapter<TModel> {\n const apiKey = getGeminiApiKeyFromEnv()\n return createGeminiVideo(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;;;;AAkEA,SAAS,sBAAsB,OAAwC;AACrE,MAAI,OAAO,MAAM,YAAY,YAAY,MAAM,QAAQ,SAAS,GAAG;AACjE,WAAO,MAAM;AAAA,EACf;AACA,SAAO,KAAK,UAAU,KAAK;AAC7B;AAcA,eAAe,oBACb,MACA,eACgB;AAChB,MAAI,KAAK,OAAO,SAAS,QAAQ;AAC/B,WAAO;AAAA,MACL,YAAY,KAAK,OAAO;AAAA,MACxB,UAAU,KAAK,OAAO,YAAY;AAAA,IAAA;AAAA,EAEtC;AACA,QAAM,MAAM,KAAK,OAAO;AACxB,MAAI,IAAI,WAAW,OAAO,GAAG;AAC3B,WAAO;AAAA,MACL,QAAQ;AAAA,MACR,GAAI,KAAK,OAAO,YAAY,EAAE,UAAU,KAAK,OAAO,SAAA;AAAA,IAAS;AAAA,EAEjE;AACA,MAAI,IAAI,WAAW,OAAO,GAAG;AAC3B,UAAM,QAAQ,IAAI,MAAM,iCAAiC;AACzD,QAAI,CAAC,SAAS,CAAC,MAAM,CAAC,GAAG;AACvB,YAAM,IAAI;AAAA,QACR;AAAA,MAAA;AAAA,IAEJ;AACA,WAAO;AAAA,MACL,YAAY,MAAM,CAAC,KAAK;AAAA,MACxB,UAAU,MAAM,CAAC,KAAK,KAAK,OAAO,YAAY;AAAA,IAAA;AAAA,EAElD;AACA,MAAI,CAAC,eAAe;AAClB,UAAM,IAAI;AAAA,MACR,gUAI+C,GAAG;AAAA,IAAA;AAAA,EAEtD;AACA,QAAM,WAAW,MAAM,MAAM,GAAG;AAChC,MAAI,CAAC,SAAS,IAAI;AAChB,UAAM,IAAI;AAAA,MACR,gCAAgC,SAAS,MAAM,IAAI,SAAS,UAAU,MAAM,GAAG;AAAA,IAAA;AAAA,EAEnF;AACA,QAAM,OAAO,MAAM,SAAS,KAAA;AAC5B,QAAM,SAAS,MAAM,KAAK,YAAA;AAC1B,SAAO;AAAA,IACL,YAAY,oBAAoB,MAAM;AAAA,IACtC,UAAU,KAAK,OAAO,YAAY,KAAK,QAAQ;AAAA,EAAA;AAEnD;AAOA,SAAS,+BACP,MACoB;AACpB,QAAM,WAAW,KAAK,OAAO;AAC7B,MAAI,KAAK,SAAS,SAAS;AACzB,WAAO,KAAK,OAAO,SAAS,SACxB,EAAE,MAAM,SAAS,MAAM,KAAK,OAAO,OAAO,WAAW,SAAA,IACrD,EAAE,MAAM,SAAS,KAAK,KAAK,OAAO,OAAO,WAAW,SAAA;AAAA,EAC1D;AACA,SAAO,KAAK,OAAO,SAAS,SACxB,EAAE,MAAM,SAAS,MAAM,KAAK,OAAO,OAAO,WAAW,SAAA,IACrD,EAAE,MAAM,SAAS,KAAK,KAAK,OAAO,OAAO,WAAW,SAAA;AAC1D;AAQA,SAAS,wBACP,aAC+D;AAC/D,QAAM,SAAS,YAAY;AAC3B,MAAI,WAAW,OAAO,QAAQ,OAAO,MAAM;AACzC,WAAO;AAAA,MACL,MAAM,OAAO;AAAA,MACb,KAAK,OAAO;AAAA,MACZ,UAAU,OAAO,aAAa;AAAA,IAAA;AAAA,EAElC;AACA,QAAM,QAAQ,YAAY,SAAS,CAAA;AACnC,WAAS,IAAI,MAAM,SAAS,GAAG,KAAK,GAAG,KAAK;AAC1C,UAAM,OAAO,MAAM,CAAC;AACpB,QAAI,MAAM,SAAS,eAAgB;AACnC,eAAW,SAAS,KAAK,WAAW,CAAA,GAAI;AACtC,UAAI,MAAM,SAAS,YAAY,MAAM,QAAQ,MAAM,MAAM;AACvD,eAAO;AAAA,UACL,MAAM,MAAM;AAAA,UACZ,KAAK,MAAM;AAAA,UACX,UAAU,MAAM,aAAa;AAAA,QAAA;AAAA,MAEjC;AAAA,IACF;AAAA,EACF;AACA,SAAO;AACT;AAOA,SAAS,6BACP,OACwB;AACxB,MAAI,CAAC,MAAO,QAAO;AACnB,QAAM,cAAc,MAAM,2BAA2B;AAAA,IACnD,CAAC,UAAU,MAAM,aAAa;AAAA,EAAA,GAC7B;AACH,QAAM,eAAe,MAAM,sBAAsB;AACjD,QAAM,mBAAmB,MAAM,uBAAuB,eAAe;AACrE,SAAO;AAAA,IACL;AAAA,IACA;AAAA,IACA,aAAa,MAAM,gBAAgB,eAAe;AAAA,EAAA;AAEtD;AAiCO,MAAM,2BAEH,iBAOR;AAAA,EACS,OAAO;AAAA,EAEN;AAAA,EACO;AAAA,EAEjB,YAAY,QAA2B,OAAe;AACpD,UAAM,CAAA,GAAI,KAAK;AACf,SAAK,SAAS,mBAAmB,MAAM;AACvC,SAAK,gBAAgB,OAAO,iBAAiB;AAAA,EAC/C;AAAA,EAEA,MAAM,eACJ,SAKyB;AACzB,UAAM,EAAE,QAAQ,MAAM,UAAU,WAAW;AAE3C,WAAO;AAAA,MACL,kCAAkC,KAAK,IAAI,UAAU,KAAK,KAAK,SAAS,QAAQ,SAAS,aAAa,YAAY,SAAS;AAAA,MAC3H,EAAE,UAAU,KAAK,MAAM,OAAO,KAAK,MAAA;AAAA,IAAM;AAG3C,QAAI,yBAAyB,KAAK,KAAK,GAAG;AACxC,aAAO,MAAM,KAAK,2BAA2B,OAAO;AAAA,IACtD;AACA,UAAM,eAAe,QAAQ;AAI7B,QAAI;AACF,YAAM,WAAW,mBAAmB,MAAM;AAE1C,UAAI,SAAS,OAAO,SAAS,GAAG;AAC9B,cAAM,IAAI;AAAA,UACR,GAAG,KAAK,IAAI,+DAA+D,KAAK,KAAK;AAAA,QAAA;AAAA,MAEzF;AACA,UAAI,SAAS,OAAO,SAAS,GAAG;AAC9B,cAAM,IAAI;AAAA,UACR,GAAG,KAAK,IAAI,+DAA+D,KAAK,KAAK;AAAA,QAAA;AAAA,MAEzF;AAEA,YAAM,EAAE,OAAO,WAAW,gBAAA,IAAoB,MAAM,KAAK;AAAA,QACvD,SAAS;AAAA,MAAA;AAGX,YAAM,SAA+B;AAAA,QACnC,GAAG;AAAA,QACH,GAAI,SAAS,UAAa,EAAE,aAAa,KAAA;AAAA,QACzC,GAAI,aAAa,UAAa,EAAE,iBAAiB,SAAA;AAAA,QACjD,GAAI,aAAa,EAAE,UAAA;AAAA,QACnB,GAAI,gBAAgB,SAAS,KAAK,EAAE,gBAAA;AAAA,MAAgB;AAGtD,YAAM,YAAY,MAAM,KAAK,OAAO,OAAO,eAAe;AAAA,QACxD,OAAO,KAAK;AAAA,QACZ,QAAQ,SAAS;AAAA,QACjB,GAAI,SAAS,EAAE,MAAA;AAAA,QACf;AAAA,MAAA,CACD;AAED,UAAI,CAAC,UAAU,MAAM;AACnB,cAAM,IAAI;AAAA,UACR;AAAA,QAAA;AAAA,MAEJ;AAEA,aAAO,EAAE,OAAO,UAAU,MAAM,OAAO,KAAK,MAAA;AAAA,IAC9C,SAAS,OAAO;AACd,aAAO,OAAO,GAAG,KAAK,IAAI,yBAAyB;AAAA,QACjD;AAAA,QACA,QAAQ,GAAG,KAAK,IAAI;AAAA,MAAA,CACrB;AACD,YAAM;AAAA,IACR;AAAA,EACF;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAOA,MAAc,2BACZ,SAKyB;AACzB,UAAM,EAAE,QAAQ,MAAM,UAAU,WAAW;AAC3C,UAAM,eAAe,QAAQ;AAI7B,QAAI;AACF,YAAM,WAAW,mBAAmB,MAAM;AAE1C,UAAI,SAAS,OAAO,SAAS,GAAG;AAC9B,cAAM,IAAI;AAAA,UACR,GAAG,KAAK,IAAI,+DAA+D,KAAK,KAAK;AAAA,QAAA;AAAA,MAEzF;AAEA,YAAM,UAAqC;AAAA,QACzC,GAAG,SAAS,OAAO,IAAI,8BAA8B;AAAA,QACrD,GAAG,SAAS,OAAO,IAAI,8BAA8B;AAAA,MAAA;AAEvD,UAAI,SAAS,MAAM;AACjB,gBAAQ,KAAK,EAAE,MAAM,QAAQ,MAAM,SAAS,MAAM;AAAA,MACpD;AACA,UAAI,QAAQ,WAAW,GAAG;AACxB,cAAM,IAAI;AAAA,UACR,GAAG,KAAK,IAAI,mEAAmE,KAAK,KAAK;AAAA,QAAA;AAAA,MAE7F;AAKA,YAAM,YAAY,KAAK,mBAAA;AACvB,UACE,aAAa,UACb,UAAU,SAAS,YAClB,WAAW,UAAU,OAAO,WAAW,UAAU,MAClD;AACA,cAAM,IAAI;AAAA,UACR,GAAG,KAAK,IAAI,6BAA6B,QAAQ,oBAAoB,UAAU,GAAG,IAAI,UAAU,GAAG,wBAAwB,KAAK,KAAK;AAAA,QAAA;AAAA,MAEzI;AAMA,YAAM,iBACJ,SAAS,UAAa,aAAa,SAC/B;AAAA,QACE,iBAAiB;AAAA,UACf,MAAM;AAAA,UACN,GAAI,SAAS,UAAa,EAAE,cAAc,KAAA;AAAA,UAC1C,GAAI,aAAa,UAAa,EAAE,UAAU,GAAG,QAAQ,IAAA;AAAA,QAAI;AAAA,MAC3D,IAEF,CAAA;AAEN,YAAM,cAAc,MAAM,KAAK,OAAO,aAAa,OAAO;AAAA,QACxD,GAAG;AAAA,QACH,OAAO,KAAK;AAAA,QACZ,OAAO,CAAC,EAAE,MAAM,cAAc,SAAS;AAAA,QACvC,qBAAqB,CAAC,OAAO;AAAA,QAC7B,YAAY;AAAA,QACZ,GAAG;AAAA,MAAA,CACJ;AAED,UAAI,CAAC,YAAY,IAAI;AACnB,cAAM,IAAI;AAAA,UACR;AAAA,QAAA;AAAA,MAEJ;AAEA,aAAO,EAAE,OAAO,YAAY,IAAI,OAAO,KAAK,MAAA;AAAA,IAC9C,SAAS,OAAO;AACd,aAAO,OAAO,GAAG,KAAK,IAAI,yBAAyB;AAAA,QACjD;AAAA,QACA,QAAQ,GAAG,KAAK,IAAI;AAAA,MAAA,CACrB;AACD,YAAM;AAAA,IACR;AAAA,EACF;AAAA;AAAA;AAAA;AAAA,EAKA,MAAc,gBACZ,OAKC;AACD,QAAI;AACJ,QAAI;AACJ,UAAM,kBAAwD,CAAA;AAE9D,eAAW,QAAQ,OAAO;AACxB,YAAM,OAAO,KAAK,UAAU;AAC5B,cAAQ,MAAA;AAAA,QACN,KAAK,aAAa;AAChB,cAAI,WAAW;AACb,kBAAM,IAAI;AAAA,cACR,GAAG,KAAK,IAAI;AAAA,YAAA;AAAA,UAEhB;AACA,sBAAY,MAAM,oBAAoB,MAAM,KAAK,aAAa;AAC9D;AAAA,QACF;AAAA,QACA,KAAK;AAAA,QACL,KAAK,aAAa;AAChB,0BAAgB,KAAK;AAAA,YACnB,OAAO,MAAM,oBAAoB,MAAM,KAAK,aAAa;AAAA,YACzD,eAAe,6BAA6B;AAAA,UAAA,CAC7C;AACD;AAAA,QACF;AAAA,QACA,KAAK;AAAA,QACL,KAAK,QAAW;AACd,cAAI,OAAO;AACT,kBAAM,IAAI;AAAA,cACR,GAAG,KAAK,IAAI;AAAA,YAAA;AAAA,UAEhB;AACA,kBAAQ,MAAM,oBAAoB,MAAM,KAAK,aAAa;AAC1D;AAAA,QACF;AAAA,QACA,KAAK;AAAA,QACL,KAAK;AACH,gBAAM,IAAI;AAAA,YACR,GAAG,KAAK,IAAI,6BAA6B,IAAI;AAAA,UAAA;AAAA,MAC/C;AAAA,IAEN;AAEA,WAAO,EAAE,OAAO,WAAW,gBAAA;AAAA,EAC7B;AAAA,EAEA,MAAM,eAAe,OAA2C;AAC9D,QAAI,yBAAyB,KAAK,KAAK,GAAG;AACxC,aAAO,MAAM,KAAK,2BAA2B,KAAK;AAAA,IACpD;AACA,UAAM,YAAY,MAAM,KAAK,aAAa,KAAK;AAE/C,QAAI,CAAC,UAAU,MAAM;AACnB,aAAO,EAAE,OAAO,QAAQ,aAAA;AAAA,IAC1B;AAEA,QAAI,UAAU,OAAO;AACnB,aAAO;AAAA,QACL;AAAA,QACA,QAAQ;AAAA,QACR,OAAO,sBAAsB,UAAU,KAAK;AAAA,MAAA;AAAA,IAEhD;AAKA,UAAM,SAAS,UAAU,UAAU,mBAAmB,CAAA;AACtD,QAAI,OAAO,WAAW,GAAG;AACvB,YAAM,UAAU,UAAU,UAAU;AACpC,aAAO;AAAA,QACL;AAAA,QACA,QAAQ;AAAA,QACR,OAAO,SAAS,SACZ,yCAAyC,QAAQ,KAAK,IAAI,CAAC,KAC3D;AAAA,MAAA;AAAA,IAER;AAEA,WAAO,EAAE,OAAO,QAAQ,YAAA;AAAA,EAC1B;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAYA,MAAc,2BACZ,OAC4B;AAC5B,UAAM,cAAc,MAAM,KAAK,eAAe,KAAK;AACnD,UAAM,SAAS,YAAY;AAE3B,QAAI,WAAW,eAAe;AAC5B,aAAO,EAAE,OAAO,QAAQ,aAAA;AAAA,IAC1B;AACA,QAAI,WAAW,mBAAmB;AAChC,aAAO;AAAA,QACL;AAAA,QACA,QAAQ;AAAA,QACR,OACE;AAAA,MAAA;AAAA,IAEN;AACA,QAAI,WAAW,aAAa;AAC1B,UAAI,CAAC,wBAAwB,WAAW,GAAG;AACzC,eAAO;AAAA,UACL;AAAA,UACA,QAAQ;AAAA,UACR,OACE;AAAA,QAAA;AAAA,MAEN;AACA,aAAO,EAAE,OAAO,QAAQ,YAAA;AAAA,IAC1B;AACA,WAAO;AAAA,MACL;AAAA,MACA,QAAQ;AAAA,MACR,OAAO,mDAAmD,MAAM;AAAA,IAAA;AAAA,EAEpE;AAAA,EAEA,MAAM,YAAY,OAAwC;AACxD,QAAI,yBAAyB,KAAK,KAAK,GAAG;AACxC,aAAO,MAAM,KAAK,wBAAwB,KAAK;AAAA,IACjD;AACA,UAAM,YAAY,MAAM,KAAK,aAAa,KAAK;AAE/C,QAAI,CAAC,UAAU,MAAM;AACnB,YAAM,IAAI;AAAA,QACR,uDAAuD,KAAK;AAAA,MAAA;AAAA,IAEhE;AAEA,QAAI,UAAU,OAAO;AACnB,YAAM,IAAI;AAAA,QACR,4BAA4B,sBAAsB,UAAU,KAAK,CAAC;AAAA,MAAA;AAAA,IAEtE;AAEA,UAAM,MAAM,UAAU,UAAU,kBAAkB,CAAC,GAAG,OAAO;AAC7D,QAAI,CAAC,KAAK;AACR,YAAM,UAAU,UAAU,UAAU;AACpC,YAAM,IAAI;AAAA,QACR,SAAS,SACL,yCAAyC,QAAQ,KAAK,IAAI,CAAC,KAC3D,sDAAsD,KAAK;AAAA,MAAA;AAAA,IAEnE;AAEA,WAAO,EAAE,OAAO,KAAK,IAAA;AAAA,EACvB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EASA,MAAc,wBACZ,OACyB;AACzB,UAAM,cAAc,MAAM,KAAK,eAAe,KAAK;AACnD,UAAM,SAAS,YAAY;AAE3B,QAAI,WAAW,eAAe;AAC5B,YAAM,IAAI;AAAA,QACR,uDAAuD,KAAK;AAAA,MAAA;AAAA,IAEhE;AACA,QAAI,WAAW,aAAa;AAC1B,YAAM,IAAI;AAAA,QACR,uEAAuE,MAAM,cAAc,KAAK;AAAA,MAAA;AAAA,IAEpG;AAEA,UAAM,QAAQ,wBAAwB,WAAW;AACjD,QAAI,CAAC,OAAO;AACV,YAAM,IAAI;AAAA,QACR,wFAAwF,KAAK;AAAA,MAAA;AAAA,IAEjG;AAEA,UAAM,QAAQ,6BAA6B,YAAY,KAAK;AAC5D,UAAM,MAAM,MAAM,OAAO,QAAQ,MAAM,QAAQ,WAAW,MAAM,IAAI;AACpE,WAAO,EAAE,OAAO,KAAK,GAAI,SAAS,EAAE,QAAM;AAAA,EAC5C;AAAA,EAES,qBAEP;AACA,WAAO,8BAA8B,KAAK,KAAK;AAAA,EACjD;AAAA,EAES,aACP,SACoD;AACpD,WAAO,qBAAqB,SAAS,KAAK,mBAAA,CAAoB;AAAA,EAChE;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAQA,MAAc,aAAa,OAAiD;AAC1E,UAAM,YAAY,IAAI,wBAAA;AACtB,cAAU,OAAO;AACjB,WAAO,MAAM,KAAK,OAAO,WAAW,mBAAmB,EAAE,WAAW;AAAA,EACtE;AAAA;AAAA;AAAA;AAAA,EAKA,MAAc,eAAe,OAAqC;AAChE,WAAO,MAAM,KAAK,OAAO,aAAa,IAAI,KAAK;AAAA,EACjD;AACF;AAwBO,SAAS,kBACd,OACA,QACA,QAC4B;AAC5B,SAAO,IAAI,mBAAmB,EAAE,QAAQ,GAAG,OAAA,GAAU,KAAK;AAC5D;AAgCO,SAAS,YACd,OACA,QAC4B;AAC5B,QAAM,SAAS,uBAAA;AACf,SAAO,kBAAkB,OAAO,QAAQ,MAAM;AAChD;"}