@tanstack/ai-gemini 0.19.1 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/dist/esm/adapters/audio.d.ts +1 -1
  2. package/dist/esm/adapters/audio.js.map +1 -1
  3. package/dist/esm/adapters/image.d.ts +1 -1
  4. package/dist/esm/adapters/image.js +17 -39
  5. package/dist/esm/adapters/image.js.map +1 -1
  6. package/dist/esm/adapters/summarize.d.ts +1 -1
  7. package/dist/esm/adapters/summarize.js.map +1 -1
  8. package/dist/esm/adapters/text.d.ts +1 -1
  9. package/dist/esm/adapters/text.js.map +1 -1
  10. package/dist/esm/adapters/tts.d.ts +1 -1
  11. package/dist/esm/adapters/tts.js.map +1 -1
  12. package/dist/esm/adapters/video.d.ts +60 -11
  13. package/dist/esm/adapters/video.js +205 -6
  14. package/dist/esm/adapters/video.js.map +1 -1
  15. package/dist/esm/experimental/text-interactions/adapter.d.ts +1 -1
  16. package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
  17. package/dist/esm/index.d.ts +6 -3
  18. package/dist/esm/index.js +9 -3
  19. package/dist/esm/index.js.map +1 -1
  20. package/dist/esm/model-meta.d.ts +11 -3
  21. package/dist/esm/model-meta.js +9 -1
  22. package/dist/esm/model-meta.js.map +1 -1
  23. package/dist/esm/realtime/adapter.d.ts +22 -0
  24. package/dist/esm/realtime/adapter.js +233 -0
  25. package/dist/esm/realtime/adapter.js.map +1 -0
  26. package/dist/esm/realtime/client.d.ts +98 -0
  27. package/dist/esm/realtime/client.js +389 -0
  28. package/dist/esm/realtime/client.js.map +1 -0
  29. package/dist/esm/realtime/index.d.ts +3 -0
  30. package/dist/esm/realtime/token.d.ts +26 -0
  31. package/dist/esm/realtime/token.js +39 -0
  32. package/dist/esm/realtime/token.js.map +1 -0
  33. package/dist/esm/realtime/types.d.ts +51 -0
  34. package/dist/esm/realtime/utils.d.ts +40 -0
  35. package/dist/esm/realtime/utils.js +350 -0
  36. package/dist/esm/realtime/utils.js.map +1 -0
  37. package/dist/esm/video/video-provider-options.d.ts +59 -14
  38. package/dist/esm/video/video-provider-options.js +15 -2
  39. package/dist/esm/video/video-provider-options.js.map +1 -1
  40. package/package.json +4 -4
  41. package/src/adapters/audio.ts +1 -1
  42. package/src/adapters/image.ts +25 -49
  43. package/src/adapters/summarize.ts +1 -1
  44. package/src/adapters/text.ts +1 -1
  45. package/src/adapters/tts.ts +1 -1
  46. package/src/adapters/video.ts +333 -16
  47. package/src/experimental/text-interactions/adapter.ts +2 -2
  48. package/src/index.ts +20 -2
  49. package/src/model-meta.ts +45 -2
  50. package/src/realtime/adapter.ts +311 -0
  51. package/src/realtime/client.ts +547 -0
  52. package/src/realtime/index.ts +14 -0
  53. package/src/realtime/token.ts +70 -0
  54. package/src/realtime/types.ts +94 -0
  55. package/src/realtime/utils.ts +439 -0
  56. package/src/video/video-provider-options.ts +95 -15
@@ -1 +1 @@
1
- {"version":3,"file":"model-meta.js","sources":["../../src/model-meta.ts"],"sourcesContent":["import type {\n GeminiCachedContentOptions,\n GeminiCommonConfigOptions,\n GeminiSafetyOptions,\n GeminiStructuredOutputOptions,\n GeminiThinkingOptions,\n GeminiToolConfigOptions,\n} from './text/text-provider-options'\n\ninterface ModelMeta<TProviderOptions = unknown> {\n name: string\n supports: {\n input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>\n output: Array<'text' | 'image' | 'audio' | 'video'>\n capabilities?: Array<\n | 'audio_generation'\n | 'batch_api'\n | 'caching'\n | 'function_calling'\n | 'live_api'\n | 'structured_output'\n | 'thinking'\n >\n tools?: Array<\n | 'code_execution'\n | 'file_search'\n | 'google_search'\n | 'google_search_retrieval'\n | 'google_maps'\n | 'url_context'\n | 'computer_use'\n >\n }\n max_input_tokens?: number\n max_output_tokens?: number\n knowledge_cutoff?: string\n pricing?: {\n input: {\n normal: number\n cached?: number\n }\n output: {\n normal: number\n }\n }\n /**\n * Type-level description of which provider options this model supports.\n */\n providerOptions?: TProviderOptions\n}\n\nconst GEMINI_3_1_PRO = {\n name: 'gemini-3.1-pro-preview',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 2.5,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_FLASH = {\n name: 'gemini-3-flash-preview',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 0.5,\n },\n output: {\n normal: 3,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_PRO_IMAGE = {\n name: 'gemini-3-pro-image-preview',\n max_input_tokens: 65_536,\n max_output_tokens: 32_768,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image'],\n output: ['text', 'image'],\n capabilities: ['batch_api', 'structured_output', 'thinking'],\n tools: ['google_search'],\n },\n pricing: {\n input: {\n normal: 2,\n },\n output: {\n normal: 0.134,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_1_FLASH_IMAGE = {\n name: 'gemini-3.1-flash-image-preview',\n max_input_tokens: 65_536,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image'],\n output: ['text', 'image'],\n capabilities: ['batch_api', 'structured_output', 'thinking'],\n tools: ['google_search'],\n },\n pricing: {\n input: {\n normal: 0.25,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_1_FLASH_LITE_IMAGE = {\n name: 'gemini-3.1-flash-lite-image',\n max_input_tokens: 65_536,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image'],\n output: ['text', 'image'],\n capabilities: ['batch_api', 'structured_output', 'thinking'],\n tools: ['google_search'],\n },\n pricing: {\n input: {\n normal: 0.25,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_1_FLASH_LITE = {\n name: 'gemini-3.1-flash-lite',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 0.25,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_1_FLASH_LITE_PREVIEW = {\n name: 'gemini-3.1-flash-lite-preview',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 0.25,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_2_5_PRO = {\n name: 'gemini-2.5-pro',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: [\n 'code_execution',\n 'file_search',\n 'google_maps',\n 'google_search',\n 'url_context',\n ],\n },\n pricing: {\n input: {\n normal: 2.5,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_2_5_PRO_TTS = {\n name: 'gemini-2.5-pro-preview-tts',\n max_input_tokens: 8_192,\n max_output_tokens: 16_384,\n knowledge_cutoff: '2025-05-01',\n supports: {\n input: ['text'],\n output: ['audio'],\n capabilities: ['audio_generation'],\n },\n pricing: {\n input: {\n normal: 2.5,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst GEMINI_2_5_FLASH = {\n name: 'gemini-2.5-flash',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: [\n 'code_execution',\n 'file_search',\n 'google_maps',\n 'google_search',\n 'url_context',\n ],\n },\n pricing: {\n input: {\n normal: 1,\n },\n output: {\n normal: 2.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_2_5_FLASH_IMAGE = {\n name: 'gemini-2.5-flash-image',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-06-01',\n supports: {\n input: ['text', 'image'],\n output: ['text', 'image'],\n capabilities: ['batch_api', 'caching', 'structured_output'],\n tools: ['file_search'],\n },\n pricing: {\n input: {\n normal: 0.3,\n },\n output: {\n normal: 0.4,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n/**\nconst GEMINI_2_5_FLASH_LIVE = {\n name: 'gemini-2.5-flash-native-audio-preview-09-2025',\n max_input_tokens: 141_072,\n max_output_tokens: 8_192,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'audio', 'video'],\n output: ['text', 'audio'],\n capabilities: [\n 'audio_generation',\n 'file_search',\n 'function_calling',\n 'live_api',\n 'search_grounding',\n 'thinking',\n ],\n },\n pricing: {\n // todo find this info\n input: {\n normal: 0,\n },\n output: {\n normal: 0,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiGenerationConfigOptions &\n GeminiCachedContentOptions &\n GeminiThinkingOptions\n>\n*/\nconst GEMINI_2_5_FLASH_TTS = {\n name: 'gemini-2.5-flash-preview-tts',\n max_input_tokens: 8_192,\n max_output_tokens: 16_384,\n knowledge_cutoff: '2025-05-01',\n supports: {\n input: ['text'],\n output: ['audio'],\n capabilities: ['audio_generation', 'batch_api'],\n },\n pricing: {\n input: {\n normal: 1,\n },\n output: {\n normal: 2.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\n/**\n * Gemini 3.1 Flash TTS Preview - latest expressive TTS model with\n * 200+ audio tags, 70+ languages, and multi-speaker dialogue support.\n * @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview\n */\nconst GEMINI_3_1_FLASH_TTS = {\n name: 'gemini-3.1-flash-tts-preview',\n max_input_tokens: 32_768,\n max_output_tokens: 16_384,\n knowledge_cutoff: '2025-05-01',\n supports: {\n input: ['text'],\n output: ['audio'],\n capabilities: ['audio_generation', 'batch_api'],\n },\n pricing: {\n input: {\n normal: 0.5,\n },\n output: {\n normal: 10,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\n/**\n * Lyria 3 Pro Preview — Google's flagship music generation model.\n * Generates full-length songs with multiple verses, choruses, and bridges.\n * Outputs MP3 or WAV at 48 kHz stereo.\n * @see https://ai.google.dev/gemini-api/docs/models/lyria-3-pro-preview\n */\nconst LYRIA_3_PRO = {\n name: 'lyria-3-pro-preview',\n max_input_tokens: 131_072,\n supports: {\n input: ['text', 'image'],\n output: ['audio'],\n capabilities: ['audio_generation'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0,\n },\n },\n} as const satisfies ModelMeta\n\n/**\n * Lyria 3 Clip Preview — 30-second music clips in MP3 format.\n * @see https://ai.google.dev/gemini-api/docs/music-generation\n */\nconst LYRIA_3_CLIP = {\n name: 'lyria-3-clip-preview',\n max_input_tokens: 131_072,\n supports: {\n input: ['text', 'image'],\n output: ['audio'],\n capabilities: ['audio_generation'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0,\n },\n },\n} as const satisfies ModelMeta\n\nconst GEMINI_2_5_FLASH_LITE = {\n name: 'gemini-2.5-flash-lite',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'google_maps', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 0.1,\n },\n output: {\n normal: 0.4,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst IMAGEN_4_GENERATE = {\n name: 'imagen-4.0-generate-001',\n max_input_tokens: 480,\n max_output_tokens: 4,\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.4,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst IMAGEN_4_GENERATE_ULTRA = {\n name: 'imagen-4.0-ultra-generate-001',\n max_input_tokens: 480,\n max_output_tokens: 4,\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.6,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst IMAGEN_4_GENERATE_FAST = {\n name: 'imagen-4.0-fast-generate-001',\n max_input_tokens: 480,\n max_output_tokens: 4,\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.2,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\n/**\n * Veo video generation models. Pricing is per second of generated video\n * (audio+video rate where the model supports audio).\n * @experimental Veo video generation is an experimental feature and may change.\n */\nconst VEO_3_1_PREVIEW = {\n name: 'veo-3.1-generate-preview',\n max_input_tokens: 1024,\n max_output_tokens: 1,\n supports: {\n input: ['text', 'image'],\n output: ['video', 'audio'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.4,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst VEO_3_1_FAST_PREVIEW = {\n name: 'veo-3.1-fast-generate-preview',\n max_input_tokens: 1024,\n max_output_tokens: 1,\n supports: {\n input: ['text', 'image'],\n output: ['video', 'audio'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.15,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst VEO_3_1_LITE_PREVIEW = {\n name: 'veo-3.1-lite-generate-preview',\n max_input_tokens: 1024,\n max_output_tokens: 1,\n supports: {\n input: ['text', 'image'],\n output: ['video', 'audio'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.05,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst GEMINI_3_5_FLASH = {\n name: 'gemini-3.5-flash',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n supports: {\n input: ['text', 'image', 'video', 'document', 'audio'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 1.5,\n cached: 0.15,\n },\n output: {\n normal: 9,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nexport const GEMINI_MODELS = [\n GEMINI_3_5_FLASH.name,\n GEMINI_3_1_PRO.name,\n GEMINI_3_FLASH.name,\n GEMINI_3_1_FLASH_LITE.name,\n GEMINI_3_1_FLASH_LITE_PREVIEW.name,\n GEMINI_2_5_PRO.name,\n GEMINI_2_5_FLASH.name,\n GEMINI_2_5_FLASH_LITE.name,\n] as const\n\n/**\n * Gemini models that support combining `tools` + `responseSchema` in a\n * single streaming `generateContent` call (per issue #605). Per the\n * provider matrix, Gemini 3.x natively interleaves the schema-constrained\n * answer with function-calling on one pass; Gemini 2.x is unsupported /\n * brittle and keeps the engine's legacy finalization fallback.\n */\nexport const GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS = new Set<string>([\n GEMINI_3_1_PRO.name,\n GEMINI_3_FLASH.name,\n GEMINI_3_1_FLASH_LITE.name,\n GEMINI_3_1_FLASH_LITE_PREVIEW.name,\n GEMINI_3_5_FLASH.name,\n])\n\nexport type GeminiModels = (typeof GEMINI_MODELS)[number]\n\nexport type GeminiImageModels = (typeof GEMINI_IMAGE_MODELS)[number]\n\nexport const GEMINI_IMAGE_MODELS = [\n GEMINI_3_1_FLASH_IMAGE.name,\n GEMINI_3_1_FLASH_LITE_IMAGE.name,\n GEMINI_3_PRO_IMAGE.name,\n GEMINI_2_5_FLASH_IMAGE.name,\n IMAGEN_4_GENERATE.name,\n IMAGEN_4_GENERATE_FAST.name,\n IMAGEN_4_GENERATE_ULTRA.name,\n] as const\n\n/**\n * Text-to-speech models\n * @experimental Gemini TTS is an experimental feature and may change.\n */\nexport const GEMINI_TTS_MODELS = [\n GEMINI_3_1_FLASH_TTS.name,\n GEMINI_2_5_FLASH_TTS.name,\n GEMINI_2_5_PRO_TTS.name,\n] as const\n\n/**\n * Audio generation models (Lyria music generation).\n * @experimental Lyria music generation is an experimental feature and may change.\n */\nexport const GEMINI_AUDIO_MODELS = [\n LYRIA_3_PRO.name,\n LYRIA_3_CLIP.name,\n] as const\n\n/**\n * Available voice names for Gemini TTS\n * @see https://ai.google.dev/gemini-api/docs/speech-generation\n */\nexport const GEMINI_TTS_VOICES = [\n 'Zephyr',\n 'Puck',\n 'Charon',\n 'Kore',\n 'Fenrir',\n 'Leda',\n 'Orus',\n 'Aoede',\n 'Callirrhoe',\n 'Autonoe',\n 'Enceladus',\n 'Iapetus',\n 'Umbriel',\n 'Algieba',\n 'Despina',\n 'Erinome',\n 'Algenib',\n 'Rasalgethi',\n 'Laomedeia',\n 'Achernar',\n 'Alnilam',\n 'Schedar',\n 'Gacrux',\n 'Pulcherrima',\n 'Achird',\n 'Zubenelgenubi',\n 'Vindemiatrix',\n 'Sadachbia',\n 'Sadaltager',\n 'Sulafat',\n] as const\n\nexport type GeminiTTSVoice = (typeof GEMINI_TTS_VOICES)[number]\n\n/**\n * Veo video generation models.\n * @experimental Veo video generation is an experimental feature and may change.\n */\nexport const GEMINI_VIDEO_MODELS = [\n VEO_3_1_PREVIEW.name,\n VEO_3_1_FAST_PREVIEW.name,\n VEO_3_1_LITE_PREVIEW.name,\n] as const\n\n// Manual type map for per-model provider options\nexport type GeminiChatModelProviderOptionsByName = {\n // Models with thinking and structured output support\n [GEMINI_3_1_PRO.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_3_FLASH.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_3_1_FLASH_LITE.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_3_1_FLASH_LITE_PREVIEW.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_2_5_PRO.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_2_5_FLASH.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_2_5_FLASH_LITE.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_3_5_FLASH.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n}\n\n/**\n * Type-only map from chat model name to its supported tool capabilities.\n * Based on the 'supports.tools' arrays defined for each model.\n */\nexport type GeminiChatModelToolCapabilitiesByName = {\n [GEMINI_3_1_PRO.name]: typeof GEMINI_3_1_PRO.supports.tools\n [GEMINI_3_FLASH.name]: typeof GEMINI_3_FLASH.supports.tools\n [GEMINI_3_1_FLASH_LITE.name]: typeof GEMINI_3_1_FLASH_LITE.supports.tools\n [GEMINI_3_1_FLASH_LITE_PREVIEW.name]: typeof GEMINI_3_1_FLASH_LITE_PREVIEW.supports.tools\n [GEMINI_2_5_PRO.name]: typeof GEMINI_2_5_PRO.supports.tools\n [GEMINI_2_5_FLASH.name]: typeof GEMINI_2_5_FLASH.supports.tools\n [GEMINI_2_5_FLASH_LITE.name]: typeof GEMINI_2_5_FLASH_LITE.supports.tools\n [GEMINI_3_5_FLASH.name]: typeof GEMINI_3_5_FLASH.supports.tools\n}\n\n/**\n * Type-only map from chat model name to its supported input modalities.\n * Based on the 'supports.input' arrays defined for each model.\n * Note: 'document' in the model meta is mapped to 'document' modality.\n * Used by the core AI types to constrain ContentPart types based on the selected model.\n * Note: These must be inlined as readonly arrays (not typeof) because the model\n * constants are not exported and typeof references don't work in .d.ts files\n * when consumed by external packages.\n *\n * @see https://ai.google.dev/gemini-api/docs/vision\n * @see https://ai.google.dev/gemini-api/docs/audio\n * @see https://ai.google.dev/gemini-api/docs/document-processing\n */\nexport type GeminiModelInputModalitiesByName = {\n // Models with full multimodal support (text, image, audio, video, document)\n [GEMINI_3_1_PRO.name]: typeof GEMINI_3_1_PRO.supports.input\n [GEMINI_3_FLASH.name]: typeof GEMINI_3_FLASH.supports.input\n [GEMINI_3_1_FLASH_LITE.name]: typeof GEMINI_3_1_FLASH_LITE.supports.input\n [GEMINI_3_1_FLASH_LITE_PREVIEW.name]: typeof GEMINI_3_1_FLASH_LITE_PREVIEW.supports.input\n [GEMINI_2_5_PRO.name]: typeof GEMINI_2_5_PRO.supports.input\n [GEMINI_2_5_FLASH_LITE.name]: typeof GEMINI_2_5_FLASH_LITE.supports.input\n [GEMINI_3_5_FLASH.name]: typeof GEMINI_3_5_FLASH.supports.input\n\n // Models with text, image, audio, video (no document)\n [GEMINI_2_5_FLASH.name]: typeof GEMINI_2_5_FLASH.supports.input\n}\n"],"names":[],"mappings":"AAmDA,MAAM,iBAAiB;AAAA,EACrB,MAAM;AAwBR;AASA,MAAM,iBAAiB;AAAA,EACrB,MAAM;AAwBR;AASA,MAAM,qBAAqB;AAAA,EACzB,MAAM;AAkBR;AASA,MAAM,yBAAyB;AAAA,EAC7B,MAAM;AAkBR;AASA,MAAM,8BAA8B;AAAA,EAClC,MAAM;AAkBR;AASA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAwBR;AASA,MAAM,gCAAgC;AAAA,EACpC,MAAM;AAwBR;AASA,MAAM,iBAAiB;AAAA,EACrB,MAAM;AA8BR;AASA,MAAM,qBAAqB;AAAA,EACzB,MAAM;AAiBR;AAOA,MAAM,mBAAmB;AAAA,EACvB,MAAM;AA8BR;AASA,MAAM,yBAAyB;AAAA,EAC7B,MAAM;AAkBR;AAyCA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAiBR;AAYA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAiBR;AAaA,MAAM,cAAc;AAAA,EAClB,MAAM;AAeR;AAMA,MAAM,eAAe;AAAA,EACnB,MAAM;AAeR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAwBR;AASA,MAAM,oBAAoB;AAAA,EACxB,MAAM;AAeR;AAOA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAeR;AAOA,MAAM,yBAAyB;AAAA,EAC7B,MAAM;AAeR;AAYA,MAAM,kBAAkB;AAAA,EACtB,MAAM;AAeR;AAOA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAeR;AAOA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAeR;AAOA,MAAM,mBAAmB;AAAA,EACvB,MAAM;AAwBR;AASO,MAAM,gBAAgB;AAAA,EAC3B,iBAAiB;AAAA,EACjB,eAAe;AAAA,EACf,eAAe;AAAA,EACf,sBAAsB;AAAA,EACtB,8BAA8B;AAAA,EAC9B,eAAe;AAAA,EACf,iBAAiB;AAAA,EACjB,sBAAsB;AACxB;AASO,MAAM,8DAA8C,IAAY;AAAA,EACrE,eAAe;AAAA,EACf,eAAe;AAAA,EACf,sBAAsB;AAAA,EACtB,8BAA8B;AAAA,EAC9B,iBAAiB;AACnB,CAAC;AAMM,MAAM,sBAAsB;AAAA,EACjC,uBAAuB;AAAA,EACvB,4BAA4B;AAAA,EAC5B,mBAAmB;AAAA,EACnB,uBAAuB;AAAA,EACvB,kBAAkB;AAAA,EAClB,uBAAuB;AAAA,EACvB,wBAAwB;AAC1B;AAMO,MAAM,oBAAoB;AAAA,EAC/B,qBAAqB;AAAA,EACrB,qBAAqB;AAAA,EACrB,mBAAmB;AACrB;AAMO,MAAM,sBAAsB;AAAA,EACjC,YAAY;AAAA,EACZ,aAAa;AACf;AAMO,MAAM,oBAAoB;AAAA,EAC/B;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAQO,MAAM,sBAAsB;AAAA,EACjC,gBAAgB;AAAA,EAChB,qBAAqB;AAAA,EACrB,qBAAqB;AACvB;"}
1
+ {"version":3,"file":"model-meta.js","sources":["../../src/model-meta.ts"],"sourcesContent":["import type {\n GeminiCachedContentOptions,\n GeminiCommonConfigOptions,\n GeminiSafetyOptions,\n GeminiStructuredOutputOptions,\n GeminiThinkingOptions,\n GeminiToolConfigOptions,\n} from './text/text-provider-options'\n\ninterface ModelMeta<TProviderOptions = unknown> {\n name: string\n supports: {\n input: Array<'text' | 'image' | 'audio' | 'video' | 'document'>\n output: Array<'text' | 'image' | 'audio' | 'video'>\n capabilities?: Array<\n | 'audio_generation'\n | 'batch_api'\n | 'caching'\n | 'function_calling'\n | 'live_api'\n | 'structured_output'\n | 'thinking'\n >\n tools?: Array<\n | 'code_execution'\n | 'file_search'\n | 'google_search'\n | 'google_search_retrieval'\n | 'google_maps'\n | 'url_context'\n | 'computer_use'\n >\n }\n max_input_tokens?: number\n max_output_tokens?: number\n knowledge_cutoff?: string\n pricing?: {\n input: {\n normal: number\n cached?: number\n }\n output: {\n normal: number\n }\n }\n /**\n * Type-level description of which provider options this model supports.\n */\n providerOptions?: TProviderOptions\n}\n\nconst GEMINI_3_1_PRO = {\n name: 'gemini-3.1-pro-preview',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 2.5,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_FLASH = {\n name: 'gemini-3-flash-preview',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 0.5,\n },\n output: {\n normal: 3,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_PRO_IMAGE = {\n name: 'gemini-3-pro-image-preview',\n max_input_tokens: 65_536,\n max_output_tokens: 32_768,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image'],\n output: ['text', 'image'],\n capabilities: ['batch_api', 'structured_output', 'thinking'],\n tools: ['google_search'],\n },\n pricing: {\n input: {\n normal: 2,\n },\n output: {\n normal: 0.134,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_1_FLASH_IMAGE = {\n name: 'gemini-3.1-flash-image-preview',\n max_input_tokens: 65_536,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image'],\n output: ['text', 'image'],\n capabilities: ['batch_api', 'structured_output', 'thinking'],\n tools: ['google_search'],\n },\n pricing: {\n input: {\n normal: 0.25,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_1_FLASH_LITE_IMAGE = {\n name: 'gemini-3.1-flash-lite-image',\n max_input_tokens: 65_536,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image'],\n output: ['text', 'image'],\n capabilities: ['batch_api', 'structured_output', 'thinking'],\n tools: ['google_search'],\n },\n pricing: {\n input: {\n normal: 0.25,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_1_FLASH_LITE = {\n name: 'gemini-3.1-flash-lite',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 0.25,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_3_1_FLASH_LITE_PREVIEW = {\n name: 'gemini-3.1-flash-lite-preview',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 0.25,\n },\n output: {\n normal: 1.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_2_5_PRO = {\n name: 'gemini-2.5-pro',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: [\n 'code_execution',\n 'file_search',\n 'google_maps',\n 'google_search',\n 'url_context',\n ],\n },\n pricing: {\n input: {\n normal: 2.5,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_2_5_PRO_TTS = {\n name: 'gemini-2.5-pro-preview-tts',\n max_input_tokens: 8_192,\n max_output_tokens: 16_384,\n knowledge_cutoff: '2025-05-01',\n supports: {\n input: ['text'],\n output: ['audio'],\n capabilities: ['audio_generation'],\n },\n pricing: {\n input: {\n normal: 2.5,\n },\n output: {\n normal: 15,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst GEMINI_2_5_FLASH = {\n name: 'gemini-2.5-flash',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: [\n 'code_execution',\n 'file_search',\n 'google_maps',\n 'google_search',\n 'url_context',\n ],\n },\n pricing: {\n input: {\n normal: 1,\n },\n output: {\n normal: 2.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst GEMINI_2_5_FLASH_IMAGE = {\n name: 'gemini-2.5-flash-image',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-06-01',\n supports: {\n input: ['text', 'image'],\n output: ['text', 'image'],\n capabilities: ['batch_api', 'caching', 'structured_output'],\n tools: ['file_search'],\n },\n pricing: {\n input: {\n normal: 0.3,\n },\n output: {\n normal: 0.4,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n/**\nconst GEMINI_2_5_FLASH_LIVE = {\n name: 'gemini-2.5-flash-native-audio-preview-09-2025',\n max_input_tokens: 141_072,\n max_output_tokens: 8_192,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'audio', 'video'],\n output: ['text', 'audio'],\n capabilities: [\n 'audio_generation',\n 'file_search',\n 'function_calling',\n 'live_api',\n 'search_grounding',\n 'thinking',\n ],\n },\n pricing: {\n // todo find this info\n input: {\n normal: 0,\n },\n output: {\n normal: 0,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiGenerationConfigOptions &\n GeminiCachedContentOptions &\n GeminiThinkingOptions\n>\n*/\nconst GEMINI_2_5_FLASH_TTS = {\n name: 'gemini-2.5-flash-preview-tts',\n max_input_tokens: 8_192,\n max_output_tokens: 16_384,\n knowledge_cutoff: '2025-05-01',\n supports: {\n input: ['text'],\n output: ['audio'],\n capabilities: ['audio_generation', 'batch_api'],\n },\n pricing: {\n input: {\n normal: 1,\n },\n output: {\n normal: 2.5,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\n/**\n * Gemini 3.1 Flash TTS Preview - latest expressive TTS model with\n * 200+ audio tags, 70+ languages, and multi-speaker dialogue support.\n * @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview\n */\nconst GEMINI_3_1_FLASH_TTS = {\n name: 'gemini-3.1-flash-tts-preview',\n max_input_tokens: 32_768,\n max_output_tokens: 16_384,\n knowledge_cutoff: '2025-05-01',\n supports: {\n input: ['text'],\n output: ['audio'],\n capabilities: ['audio_generation', 'batch_api'],\n },\n pricing: {\n input: {\n normal: 0.5,\n },\n output: {\n normal: 10,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\n/**\n * Lyria 3 Pro Preview — Google's flagship music generation model.\n * Generates full-length songs with multiple verses, choruses, and bridges.\n * Outputs MP3 or WAV at 48 kHz stereo.\n * @see https://ai.google.dev/gemini-api/docs/models/lyria-3-pro-preview\n */\nconst LYRIA_3_PRO = {\n name: 'lyria-3-pro-preview',\n max_input_tokens: 131_072,\n supports: {\n input: ['text', 'image'],\n output: ['audio'],\n capabilities: ['audio_generation'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0,\n },\n },\n} as const satisfies ModelMeta\n\n/**\n * Lyria 3 Clip Preview — 30-second music clips in MP3 format.\n * @see https://ai.google.dev/gemini-api/docs/music-generation\n */\nconst LYRIA_3_CLIP = {\n name: 'lyria-3-clip-preview',\n max_input_tokens: 131_072,\n supports: {\n input: ['text', 'image'],\n output: ['audio'],\n capabilities: ['audio_generation'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0,\n },\n },\n} as const satisfies ModelMeta\n\nconst GEMINI_2_5_FLASH_LITE = {\n name: 'gemini-2.5-flash-lite',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n knowledge_cutoff: '2025-01-01',\n supports: {\n input: ['text', 'image', 'audio', 'video', 'document'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'google_maps', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 0.1,\n },\n output: {\n normal: 0.4,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nconst IMAGEN_4_GENERATE = {\n name: 'imagen-4.0-generate-001',\n max_input_tokens: 480,\n max_output_tokens: 4,\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.4,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst IMAGEN_4_GENERATE_ULTRA = {\n name: 'imagen-4.0-ultra-generate-001',\n max_input_tokens: 480,\n max_output_tokens: 4,\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.6,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst IMAGEN_4_GENERATE_FAST = {\n name: 'imagen-4.0-fast-generate-001',\n max_input_tokens: 480,\n max_output_tokens: 4,\n supports: {\n input: ['text'],\n output: ['image'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.2,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\n/**\n * Veo video generation models. Pricing is per second of generated video\n * (audio+video rate where the model supports audio).\n * @experimental Veo video generation is an experimental feature and may change.\n */\nconst VEO_3_1_PREVIEW = {\n name: 'veo-3.1-generate-preview',\n max_input_tokens: 1024,\n max_output_tokens: 1,\n supports: {\n input: ['text', 'image'],\n output: ['video', 'audio'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.4,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst VEO_3_1_FAST_PREVIEW = {\n name: 'veo-3.1-fast-generate-preview',\n max_input_tokens: 1024,\n max_output_tokens: 1,\n supports: {\n input: ['text', 'image'],\n output: ['video', 'audio'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.15,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst VEO_3_1_LITE_PREVIEW = {\n name: 'veo-3.1-lite-generate-preview',\n max_input_tokens: 1024,\n max_output_tokens: 1,\n supports: {\n input: ['text', 'image'],\n output: ['video', 'audio'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.05,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\n/**\n * Gemini Omni Flash — multimodal video generation with conversational\n * editing. Serves only the Interactions API (`generateContent` rejects it),\n * so it routes through the interactions-based path of the video adapter,\n * not Veo's `:predictLongRunning` flow. Pricing is per second of generated\n * video ($0.10/sec). 720p / 24 FPS, 3–10 second clips (default 10s).\n * @experimental Omni video generation is an experimental feature and may change.\n */\nconst GEMINI_OMNI_FLASH_PREVIEW = {\n name: 'gemini-omni-flash-preview',\n max_input_tokens: 1_048_576,\n max_output_tokens: 1,\n supports: {\n input: ['text', 'image', 'video'],\n output: ['video', 'audio'],\n },\n pricing: {\n input: {\n normal: 0,\n },\n output: {\n normal: 0.1,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions\n>\n\nconst GEMINI_3_5_FLASH = {\n name: 'gemini-3.5-flash',\n max_input_tokens: 1_048_576,\n max_output_tokens: 65_536,\n supports: {\n input: ['text', 'image', 'video', 'document', 'audio'],\n output: ['text'],\n capabilities: [\n 'batch_api',\n 'caching',\n 'function_calling',\n 'structured_output',\n 'thinking',\n ],\n tools: ['code_execution', 'file_search', 'google_search', 'url_context'],\n },\n pricing: {\n input: {\n normal: 1.5,\n cached: 0.15,\n },\n output: {\n normal: 9,\n },\n },\n} as const satisfies ModelMeta<\n GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n>\n\nexport const GEMINI_MODELS = [\n GEMINI_3_5_FLASH.name,\n GEMINI_3_1_PRO.name,\n GEMINI_3_FLASH.name,\n GEMINI_3_1_FLASH_LITE.name,\n GEMINI_3_1_FLASH_LITE_PREVIEW.name,\n GEMINI_2_5_PRO.name,\n GEMINI_2_5_FLASH.name,\n GEMINI_2_5_FLASH_LITE.name,\n] as const\n\n/**\n * Gemini models that support combining `tools` + `responseSchema` in a\n * single streaming `generateContent` call (per issue #605). Per the\n * provider matrix, Gemini 3.x natively interleaves the schema-constrained\n * answer with function-calling on one pass; Gemini 2.x is unsupported /\n * brittle and keeps the engine's legacy finalization fallback.\n */\nexport const GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS = new Set<string>([\n GEMINI_3_1_PRO.name,\n GEMINI_3_FLASH.name,\n GEMINI_3_1_FLASH_LITE.name,\n GEMINI_3_1_FLASH_LITE_PREVIEW.name,\n GEMINI_3_5_FLASH.name,\n])\n\nexport type GeminiModels = (typeof GEMINI_MODELS)[number]\n\nexport type GeminiImageModels = (typeof GEMINI_IMAGE_MODELS)[number]\n\nexport const GEMINI_IMAGE_MODELS = [\n GEMINI_3_1_FLASH_IMAGE.name,\n GEMINI_3_1_FLASH_LITE_IMAGE.name,\n GEMINI_3_PRO_IMAGE.name,\n GEMINI_2_5_FLASH_IMAGE.name,\n IMAGEN_4_GENERATE.name,\n IMAGEN_4_GENERATE_FAST.name,\n IMAGEN_4_GENERATE_ULTRA.name,\n] as const\n\n/**\n * Text-to-speech models\n * @experimental Gemini TTS is an experimental feature and may change.\n */\nexport const GEMINI_TTS_MODELS = [\n GEMINI_3_1_FLASH_TTS.name,\n GEMINI_2_5_FLASH_TTS.name,\n GEMINI_2_5_PRO_TTS.name,\n] as const\n\n/**\n * Audio generation models (Lyria music generation).\n * @experimental Lyria music generation is an experimental feature and may change.\n */\nexport const GEMINI_AUDIO_MODELS = [\n LYRIA_3_PRO.name,\n LYRIA_3_CLIP.name,\n] as const\n\n/**\n * Available voice names for Gemini TTS\n * @see https://ai.google.dev/gemini-api/docs/speech-generation\n */\nexport const GEMINI_TTS_VOICES = [\n 'Zephyr',\n 'Puck',\n 'Charon',\n 'Kore',\n 'Fenrir',\n 'Leda',\n 'Orus',\n 'Aoede',\n 'Callirrhoe',\n 'Autonoe',\n 'Enceladus',\n 'Iapetus',\n 'Umbriel',\n 'Algieba',\n 'Despina',\n 'Erinome',\n 'Algenib',\n 'Rasalgethi',\n 'Laomedeia',\n 'Achernar',\n 'Alnilam',\n 'Schedar',\n 'Gacrux',\n 'Pulcherrima',\n 'Achird',\n 'Zubenelgenubi',\n 'Vindemiatrix',\n 'Sadachbia',\n 'Sadaltager',\n 'Sulafat',\n] as const\n\nexport type GeminiTTSVoice = (typeof GEMINI_TTS_VOICES)[number]\n\n/**\n * Video generation models. Veo models run on the long-running\n * `:predictLongRunning` flow; Gemini Omni Flash runs on the Interactions\n * API — the video adapter routes by model.\n * @experimental Video generation is an experimental feature and may change.\n */\nexport const GEMINI_VIDEO_MODELS = [\n VEO_3_1_PREVIEW.name,\n VEO_3_1_FAST_PREVIEW.name,\n VEO_3_1_LITE_PREVIEW.name,\n GEMINI_OMNI_FLASH_PREVIEW.name,\n] as const\n\n/**\n * Video models served by the Interactions API rather than Veo's\n * `:predictLongRunning` operations flow.\n * @experimental Omni video generation is an experimental feature and may change.\n */\nexport const GEMINI_INTERACTIONS_VIDEO_MODELS = [\n GEMINI_OMNI_FLASH_PREVIEW.name,\n] as const\n\n// Manual type map for per-model provider options\nexport type GeminiChatModelProviderOptionsByName = {\n // Models with thinking and structured output support\n [GEMINI_3_1_PRO.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_3_FLASH.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_3_1_FLASH_LITE.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_3_1_FLASH_LITE_PREVIEW.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_2_5_PRO.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_2_5_FLASH.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_2_5_FLASH_LITE.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n [GEMINI_3_5_FLASH.name]: GeminiToolConfigOptions &\n GeminiSafetyOptions &\n GeminiCommonConfigOptions &\n GeminiCachedContentOptions &\n GeminiStructuredOutputOptions &\n GeminiThinkingOptions\n}\n\n/**\n * Type-only map from chat model name to its supported tool capabilities.\n * Based on the 'supports.tools' arrays defined for each model.\n */\nexport type GeminiChatModelToolCapabilitiesByName = {\n [GEMINI_3_1_PRO.name]: typeof GEMINI_3_1_PRO.supports.tools\n [GEMINI_3_FLASH.name]: typeof GEMINI_3_FLASH.supports.tools\n [GEMINI_3_1_FLASH_LITE.name]: typeof GEMINI_3_1_FLASH_LITE.supports.tools\n [GEMINI_3_1_FLASH_LITE_PREVIEW.name]: typeof GEMINI_3_1_FLASH_LITE_PREVIEW.supports.tools\n [GEMINI_2_5_PRO.name]: typeof GEMINI_2_5_PRO.supports.tools\n [GEMINI_2_5_FLASH.name]: typeof GEMINI_2_5_FLASH.supports.tools\n [GEMINI_2_5_FLASH_LITE.name]: typeof GEMINI_2_5_FLASH_LITE.supports.tools\n [GEMINI_3_5_FLASH.name]: typeof GEMINI_3_5_FLASH.supports.tools\n}\n\n/**\n * Type-only map from chat model name to its supported input modalities.\n * Based on the 'supports.input' arrays defined for each model.\n * Note: 'document' in the model meta is mapped to 'document' modality.\n * Used by the core AI types to constrain ContentPart types based on the selected model.\n * Note: These must be inlined as readonly arrays (not typeof) because the model\n * constants are not exported and typeof references don't work in .d.ts files\n * when consumed by external packages.\n *\n * @see https://ai.google.dev/gemini-api/docs/vision\n * @see https://ai.google.dev/gemini-api/docs/audio\n * @see https://ai.google.dev/gemini-api/docs/document-processing\n */\nexport type GeminiModelInputModalitiesByName = {\n // Models with full multimodal support (text, image, audio, video, document)\n [GEMINI_3_1_PRO.name]: typeof GEMINI_3_1_PRO.supports.input\n [GEMINI_3_FLASH.name]: typeof GEMINI_3_FLASH.supports.input\n [GEMINI_3_1_FLASH_LITE.name]: typeof GEMINI_3_1_FLASH_LITE.supports.input\n [GEMINI_3_1_FLASH_LITE_PREVIEW.name]: typeof GEMINI_3_1_FLASH_LITE_PREVIEW.supports.input\n [GEMINI_2_5_PRO.name]: typeof GEMINI_2_5_PRO.supports.input\n [GEMINI_2_5_FLASH_LITE.name]: typeof GEMINI_2_5_FLASH_LITE.supports.input\n [GEMINI_3_5_FLASH.name]: typeof GEMINI_3_5_FLASH.supports.input\n\n // Models with text, image, audio, video (no document)\n [GEMINI_2_5_FLASH.name]: typeof GEMINI_2_5_FLASH.supports.input\n}\n"],"names":[],"mappings":"AAmDA,MAAM,iBAAiB;AAAA,EACrB,MAAM;AAwBR;AASA,MAAM,iBAAiB;AAAA,EACrB,MAAM;AAwBR;AASA,MAAM,qBAAqB;AAAA,EACzB,MAAM;AAkBR;AASA,MAAM,yBAAyB;AAAA,EAC7B,MAAM;AAkBR;AASA,MAAM,8BAA8B;AAAA,EAClC,MAAM;AAkBR;AASA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAwBR;AASA,MAAM,gCAAgC;AAAA,EACpC,MAAM;AAwBR;AASA,MAAM,iBAAiB;AAAA,EACrB,MAAM;AA8BR;AASA,MAAM,qBAAqB;AAAA,EACzB,MAAM;AAiBR;AAOA,MAAM,mBAAmB;AAAA,EACvB,MAAM;AA8BR;AASA,MAAM,yBAAyB;AAAA,EAC7B,MAAM;AAkBR;AAyCA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAiBR;AAYA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAiBR;AAaA,MAAM,cAAc;AAAA,EAClB,MAAM;AAeR;AAMA,MAAM,eAAe;AAAA,EACnB,MAAM;AAeR;AAEA,MAAM,wBAAwB;AAAA,EAC5B,MAAM;AAwBR;AASA,MAAM,oBAAoB;AAAA,EACxB,MAAM;AAeR;AAOA,MAAM,0BAA0B;AAAA,EAC9B,MAAM;AAeR;AAOA,MAAM,yBAAyB;AAAA,EAC7B,MAAM;AAeR;AAYA,MAAM,kBAAkB;AAAA,EACtB,MAAM;AAeR;AAOA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAeR;AAOA,MAAM,uBAAuB;AAAA,EAC3B,MAAM;AAeR;AAeA,MAAM,4BAA4B;AAAA,EAChC,MAAM;AAeR;AAOA,MAAM,mBAAmB;AAAA,EACvB,MAAM;AAwBR;AASO,MAAM,gBAAgB;AAAA,EAC3B,iBAAiB;AAAA,EACjB,eAAe;AAAA,EACf,eAAe;AAAA,EACf,sBAAsB;AAAA,EACtB,8BAA8B;AAAA,EAC9B,eAAe;AAAA,EACf,iBAAiB;AAAA,EACjB,sBAAsB;AACxB;AASO,MAAM,8DAA8C,IAAY;AAAA,EACrE,eAAe;AAAA,EACf,eAAe;AAAA,EACf,sBAAsB;AAAA,EACtB,8BAA8B;AAAA,EAC9B,iBAAiB;AACnB,CAAC;AAMM,MAAM,sBAAsB;AAAA,EACjC,uBAAuB;AAAA,EACvB,4BAA4B;AAAA,EAC5B,mBAAmB;AAAA,EACnB,uBAAuB;AAAA,EACvB,kBAAkB;AAAA,EAClB,uBAAuB;AAAA,EACvB,wBAAwB;AAC1B;AAMO,MAAM,oBAAoB;AAAA,EAC/B,qBAAqB;AAAA,EACrB,qBAAqB;AAAA,EACrB,mBAAmB;AACrB;AAMO,MAAM,sBAAsB;AAAA,EACjC,YAAY;AAAA,EACZ,aAAa;AACf;AAMO,MAAM,oBAAoB;AAAA,EAC/B;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAUO,MAAM,sBAAsB;AAAA,EACjC,gBAAgB;AAAA,EAChB,qBAAqB;AAAA,EACrB,qBAAqB;AAAA,EACrB,0BAA0B;AAC5B;AAOO,MAAM,mCAAmC;AAAA,EAC9C,0BAA0B;AAC5B;"}
@@ -0,0 +1,22 @@
1
+ import { RealtimeAdapter } from '@tanstack/ai';
2
+ import { GeminiRealtimeOptions } from './types.js';
3
+ /**
4
+ * Creates a Gemini realtime adapter for client-side use.
5
+ *
6
+ * @param options - Optional configuration
7
+ * @returns A RealtimeAdapter for use with RealtimeClient
8
+ *
9
+ * @example
10
+ * ```typescript
11
+ * import { RealtimeClient } from '@tanstack/ai-client'
12
+ * import { geminiRealtime } from '@tanstack/ai-gemini'
13
+ *
14
+ * const client = new RealtimeClient({
15
+ * getToken: () => fetch('/api/realtime-token').then(r => r.json()),
16
+ * adapter: geminiRealtime(),
17
+ * onGoAway: () => client.updateSession({ ... }) // Resume session with new config (available only for Gemini Live adapter)
18
+ * })
19
+ *
20
+ * ```
21
+ */
22
+ export declare function geminiRealtime(options?: GeminiRealtimeOptions): RealtimeAdapter;
@@ -0,0 +1,233 @@
1
+ import { createRealtimeEventEmitter } from "@tanstack/ai";
2
+ import { AudioPlayer, base64ToArrayBuffer, AudioStreamer } from "./utils.js";
3
+ import { GeminiLiveClient } from "./client.js";
4
+ function geminiRealtime(options = {}) {
5
+ return {
6
+ provider: "gemini",
7
+ connect(token, clientTools) {
8
+ return createWebSocketConnection(token, options.model, clientTools);
9
+ }
10
+ };
11
+ }
12
+ async function createWebSocketConnection(token, model = "gemini-3.1-flash-live-preview", tools) {
13
+ const { emit, on: realtimeEventEmitterOn } = createRealtimeEventEmitter();
14
+ let currentMode = "idle";
15
+ let currentMessageId = null;
16
+ let messageIdCounter = 0;
17
+ function generateMessageId() {
18
+ return `gemini-msg-${Date.now()}-${++messageIdCounter}`;
19
+ }
20
+ const client = new GeminiLiveClient(token.token, model, tools);
21
+ let message = {
22
+ id: "",
23
+ role: "assistant",
24
+ timestamp: 0,
25
+ parts: []
26
+ };
27
+ let pendingAssistantResponse = "";
28
+ client.onClose = () => {
29
+ emit("status_change", { status: "idle" });
30
+ emit("mode_change", { mode: "idle" });
31
+ };
32
+ client.onError = (error) => {
33
+ emit("error", { error });
34
+ emit("status_change", { status: "error" });
35
+ emit("mode_change", { mode: "idle" });
36
+ };
37
+ client.onReceiveResponse = (response) => {
38
+ switch (response.type) {
39
+ case "text":
40
+ message.parts.push({
41
+ type: "text",
42
+ content: response.data
43
+ });
44
+ break;
45
+ case "audio": {
46
+ const pcm = base64ToArrayBuffer(response.data.audioData);
47
+ message.parts.push({
48
+ type: "audio",
49
+ transcript: response.data.transcript,
50
+ audioData: pcm
51
+ });
52
+ if (currentMode !== "speaking") {
53
+ currentMode = "speaking";
54
+ emit("mode_change", { mode: "speaking" });
55
+ }
56
+ audioPlayer.play(pcm).catch((error) => emit("error", { error }));
57
+ break;
58
+ }
59
+ case "go_away":
60
+ emit("go_away", { timeLeft: response.data.timeLeft });
61
+ break;
62
+ case "usage_metadata":
63
+ emit("usage", {
64
+ completionTokens: response.data.responseTokenCount ?? 0,
65
+ promptTokens: response.data.promptTokenCount ?? 0,
66
+ totalTokens: response.data.totalTokenCount ?? 0
67
+ });
68
+ break;
69
+ case "input_transcription":
70
+ if (response.data.finished && currentMode !== "thinking") {
71
+ currentMode = "thinking";
72
+ emit("mode_change", { mode: "thinking" });
73
+ }
74
+ emit("transcript", {
75
+ isFinal: response.data.finished,
76
+ transcript: response.data.text,
77
+ role: "user"
78
+ });
79
+ break;
80
+ case "output_transcription":
81
+ pendingAssistantResponse += response.data.text;
82
+ emit("transcript", {
83
+ isFinal: response.data.finished,
84
+ transcript: pendingAssistantResponse,
85
+ role: "assistant"
86
+ });
87
+ break;
88
+ case "interrupted":
89
+ audioPlayer.interrupt();
90
+ currentMode = "listening";
91
+ emit("mode_change", { mode: "listening" });
92
+ emit("interrupted", { messageId: currentMessageId ?? void 0 });
93
+ break;
94
+ case "tool_call":
95
+ for (const tool of response.data.functionCalls || []) {
96
+ if (tool.id && tool.name) {
97
+ emit("tool_call", {
98
+ toolCallId: tool.id,
99
+ input: tool.args,
100
+ toolName: tool.name
101
+ });
102
+ }
103
+ }
104
+ break;
105
+ case "turn_complete":
106
+ currentMessageId = generateMessageId();
107
+ message.id = currentMessageId;
108
+ message.timestamp = Date.now();
109
+ message.parts.push({
110
+ type: "text",
111
+ content: pendingAssistantResponse
112
+ });
113
+ emit("message_complete", { message });
114
+ pendingAssistantResponse = "";
115
+ message = {
116
+ id: "",
117
+ role: "assistant",
118
+ timestamp: 0,
119
+ parts: []
120
+ };
121
+ currentMode = "listening";
122
+ emit("mode_change", { mode: "listening" });
123
+ break;
124
+ case "setup_complete":
125
+ emit("status_change", { status: "connected" });
126
+ break;
127
+ case "error":
128
+ emit("error", {
129
+ error: new Error(response.data)
130
+ });
131
+ break;
132
+ }
133
+ };
134
+ await client.connect();
135
+ const audioStreamer = new AudioStreamer(client);
136
+ const audioPlayer = new AudioPlayer();
137
+ try {
138
+ await audioPlayer.init();
139
+ await audioStreamer.start();
140
+ } catch (error) {
141
+ audioStreamer.stop();
142
+ audioPlayer.destroy();
143
+ client.disconnect();
144
+ throw error;
145
+ }
146
+ const connection = {
147
+ async disconnect() {
148
+ audioStreamer.stop();
149
+ audioPlayer.destroy();
150
+ client.disconnect();
151
+ currentMode = "idle";
152
+ emit("status_change", { status: "idle" });
153
+ },
154
+ async startAudioCapture() {
155
+ audioStreamer.startAudioCapture();
156
+ currentMode = "listening";
157
+ emit("mode_change", { mode: "listening" });
158
+ },
159
+ stopAudioCapture() {
160
+ audioStreamer.stopAudioCapture();
161
+ currentMode = "idle";
162
+ emit("mode_change", { mode: "idle" });
163
+ },
164
+ sendText(text) {
165
+ client.sendTextMessage(text);
166
+ currentMode = "thinking";
167
+ emit("mode_change", { mode: "thinking" });
168
+ },
169
+ sendImage(imageData, mimeType) {
170
+ client.sendImageMessage(imageData, mimeType);
171
+ currentMode = "thinking";
172
+ emit("mode_change", { mode: "thinking" });
173
+ },
174
+ sendToolResult(callId, result) {
175
+ client.sendToolResponse([
176
+ {
177
+ id: callId,
178
+ response: {
179
+ result
180
+ }
181
+ }
182
+ ]);
183
+ },
184
+ updateSession(config) {
185
+ client.updateSession(config).catch((error) => emit("error", { error }));
186
+ emit("status_change", { status: "reconnecting" });
187
+ },
188
+ updateToken(newToken) {
189
+ client.updateToken(newToken);
190
+ emit("status_change", { status: "reconnecting" });
191
+ },
192
+ interrupt() {
193
+ audioPlayer.interrupt();
194
+ currentMode = "listening";
195
+ emit("mode_change", { mode: "listening" });
196
+ emit("interrupted", { messageId: currentMessageId ?? void 0 });
197
+ },
198
+ on: realtimeEventEmitterOn,
199
+ getAudioVisualization() {
200
+ return {
201
+ get inputLevel() {
202
+ return audioStreamer.inputLevel;
203
+ },
204
+ get outputLevel() {
205
+ return audioPlayer.outputLevel;
206
+ },
207
+ getInputFrequencyData() {
208
+ return audioStreamer.inputFrequencyData;
209
+ },
210
+ getOutputFrequencyData() {
211
+ return audioPlayer.outputFrequencyData;
212
+ },
213
+ getInputTimeDomainData() {
214
+ return audioStreamer.inputTimeDomainData;
215
+ },
216
+ getOutputTimeDomainData() {
217
+ return audioPlayer.outputTimeDomainData;
218
+ },
219
+ get inputSampleRate() {
220
+ return audioStreamer.inputSampleRate;
221
+ },
222
+ get outputSampleRate() {
223
+ return audioPlayer.outputSampleRate;
224
+ }
225
+ };
226
+ }
227
+ };
228
+ return connection;
229
+ }
230
+ export {
231
+ geminiRealtime
232
+ };
233
+ //# sourceMappingURL=adapter.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"adapter.js","sources":["../../../src/realtime/adapter.ts"],"sourcesContent":["import { createRealtimeEventEmitter } from '@tanstack/ai'\nimport { AudioPlayer, AudioStreamer, base64ToArrayBuffer } from './utils'\nimport { GeminiLiveClient } from './client'\nimport type { LiveResponse } from './client'\nimport type {\n AnyClientTool,\n AudioVisualization,\n RealtimeAdapter,\n RealtimeConnection,\n RealtimeMessage,\n RealtimeMode,\n RealtimeToken,\n} from '@tanstack/ai'\nimport type { GeminiRealtimeModel, GeminiRealtimeOptions } from './types'\n\n/**\n * Creates a Gemini realtime adapter for client-side use.\n *\n * @param options - Optional configuration\n * @returns A RealtimeAdapter for use with RealtimeClient\n *\n * @example\n * ```typescript\n * import { RealtimeClient } from '@tanstack/ai-client'\n * import { geminiRealtime } from '@tanstack/ai-gemini'\n *\n * const client = new RealtimeClient({\n * getToken: () => fetch('/api/realtime-token').then(r => r.json()),\n * adapter: geminiRealtime(),\n * onGoAway: () => client.updateSession({ ... }) // Resume session with new config (available only for Gemini Live adapter)\n * })\n *\n * ```\n */\nexport function geminiRealtime(\n options: GeminiRealtimeOptions = {},\n): RealtimeAdapter {\n return {\n provider: 'gemini',\n\n connect(\n token: RealtimeToken,\n clientTools?: ReadonlyArray<AnyClientTool>,\n ): Promise<RealtimeConnection> {\n return createWebSocketConnection(token, options.model, clientTools)\n },\n }\n}\n\n/**\n * Creates a WebSocket connection to Gemini's realtime API\n */\nasync function createWebSocketConnection(\n token: RealtimeToken,\n model: GeminiRealtimeModel = 'gemini-3.1-flash-live-preview',\n tools?: ReadonlyArray<AnyClientTool>,\n): Promise<RealtimeConnection> {\n const { emit, on: realtimeEventEmitterOn } = createRealtimeEventEmitter()\n\n // Current state\n let currentMode: RealtimeMode = 'idle'\n let currentMessageId: string | null = null\n let messageIdCounter = 0\n\n function generateMessageId(): string {\n return `gemini-msg-${Date.now()}-${++messageIdCounter}`\n }\n\n const client = new GeminiLiveClient(token.token, model, tools)\n\n let message: RealtimeMessage = {\n id: '',\n role: 'assistant',\n timestamp: 0,\n parts: [],\n }\n\n let pendingAssistantResponse = ''\n\n client.onClose = () => {\n emit('status_change', { status: 'idle' })\n emit('mode_change', { mode: 'idle' })\n }\n\n client.onError = (error) => {\n emit('error', { error })\n emit('status_change', { status: 'error' })\n emit('mode_change', { mode: 'idle' })\n }\n\n client.onReceiveResponse = (response: LiveResponse) => {\n switch (response.type) {\n case 'text':\n message.parts.push({\n type: 'text',\n content: response.data,\n })\n break\n case 'audio': {\n // Decode once, then reuse for both the stored message part and playback.\n const pcm = base64ToArrayBuffer(response.data.audioData)\n message.parts.push({\n type: 'audio',\n transcript: response.data.transcript,\n audioData: pcm,\n })\n if (currentMode !== 'speaking') {\n currentMode = 'speaking'\n emit('mode_change', { mode: 'speaking' })\n }\n audioPlayer.play(pcm).catch((error) => emit('error', { error }))\n break\n }\n case 'go_away':\n emit('go_away', { timeLeft: response.data.timeLeft })\n break\n case 'usage_metadata':\n emit('usage', {\n completionTokens: response.data.responseTokenCount ?? 0,\n promptTokens: response.data.promptTokenCount ?? 0,\n totalTokens: response.data.totalTokenCount ?? 0,\n })\n break\n case 'input_transcription':\n if (response.data.finished && currentMode !== 'thinking') {\n currentMode = 'thinking'\n emit('mode_change', { mode: 'thinking' })\n }\n emit('transcript', {\n isFinal: response.data.finished,\n transcript: response.data.text,\n role: 'user',\n })\n break\n case 'output_transcription':\n pendingAssistantResponse += response.data.text\n emit('transcript', {\n isFinal: response.data.finished,\n transcript: pendingAssistantResponse,\n role: 'assistant',\n })\n break\n case 'interrupted':\n audioPlayer.interrupt()\n currentMode = 'listening'\n emit('mode_change', { mode: 'listening' })\n emit('interrupted', { messageId: currentMessageId ?? undefined })\n break\n case 'tool_call':\n for (const tool of response.data.functionCalls || []) {\n if (tool.id && tool.name) {\n emit('tool_call', {\n toolCallId: tool.id,\n input: tool.args,\n toolName: tool.name,\n })\n }\n }\n break\n case 'turn_complete':\n currentMessageId = generateMessageId()\n message.id = currentMessageId\n message.timestamp = Date.now()\n message.parts.push({\n type: 'text',\n content: pendingAssistantResponse,\n })\n emit('message_complete', { message })\n\n pendingAssistantResponse = ''\n message = {\n id: '',\n role: 'assistant',\n timestamp: 0,\n parts: [],\n }\n currentMode = 'listening'\n emit('mode_change', { mode: 'listening' })\n break\n case 'setup_complete':\n emit('status_change', { status: 'connected' })\n break\n case 'error':\n emit('error', {\n error: new Error(response.data),\n })\n break\n case 'thought':\n case 'session_resumption_update':\n // Handled internally by GeminiLiveClient; not surfaced to the client.\n break\n }\n }\n\n await client.connect()\n\n const audioStreamer = new AudioStreamer(client)\n const audioPlayer = new AudioPlayer()\n try {\n await audioPlayer.init()\n await audioStreamer.start()\n } catch (error) {\n // Tear down the socket + audio graph if mic/worklet setup fails (e.g. the\n // user denied microphone access) rather than leaking an open connection.\n audioStreamer.stop()\n audioPlayer.destroy()\n client.disconnect()\n throw error\n }\n\n const connection: RealtimeConnection = {\n async disconnect() {\n audioStreamer.stop()\n audioPlayer.destroy()\n client.disconnect()\n currentMode = 'idle'\n emit('status_change', { status: 'idle' })\n },\n\n async startAudioCapture() {\n // Audio capture is established during connection setup\n audioStreamer.startAudioCapture()\n currentMode = 'listening'\n emit('mode_change', { mode: 'listening' })\n },\n\n stopAudioCapture() {\n audioStreamer.stopAudioCapture()\n currentMode = 'idle'\n emit('mode_change', { mode: 'idle' })\n },\n\n sendText(text: string) {\n client.sendTextMessage(text)\n currentMode = 'thinking'\n emit('mode_change', { mode: 'thinking' })\n },\n\n sendImage(imageData: string, mimeType: string) {\n client.sendImageMessage(imageData, mimeType)\n currentMode = 'thinking'\n emit('mode_change', { mode: 'thinking' })\n },\n\n sendToolResult(callId: string, result: string) {\n client.sendToolResponse([\n {\n id: callId,\n response: {\n result,\n },\n },\n ])\n },\n\n updateSession(config) {\n client.updateSession(config).catch((error) => emit('error', { error }))\n emit('status_change', { status: 'reconnecting' })\n },\n\n updateToken(newToken) {\n client.updateToken(newToken)\n emit('status_change', { status: 'reconnecting' })\n },\n\n interrupt() {\n audioPlayer.interrupt()\n currentMode = 'listening'\n emit('mode_change', { mode: 'listening' })\n emit('interrupted', { messageId: currentMessageId ?? undefined })\n },\n on: realtimeEventEmitterOn,\n getAudioVisualization(): AudioVisualization {\n return {\n get inputLevel() {\n return audioStreamer.inputLevel\n },\n\n get outputLevel() {\n return audioPlayer.outputLevel\n },\n\n getInputFrequencyData() {\n return audioStreamer.inputFrequencyData\n },\n\n getOutputFrequencyData() {\n return audioPlayer.outputFrequencyData\n },\n\n getInputTimeDomainData() {\n return audioStreamer.inputTimeDomainData\n },\n\n getOutputTimeDomainData() {\n return audioPlayer.outputTimeDomainData\n },\n\n get inputSampleRate() {\n return audioStreamer.inputSampleRate\n },\n\n get outputSampleRate() {\n return audioPlayer.outputSampleRate\n },\n }\n },\n }\n\n return connection\n}\n"],"names":[],"mappings":";;;AAkCO,SAAS,eACd,UAAiC,IAChB;AACjB,SAAO;AAAA,IACL,UAAU;AAAA,IAEV,QACE,OACA,aAC6B;AAC7B,aAAO,0BAA0B,OAAO,QAAQ,OAAO,WAAW;AAAA,IACpE;AAAA,EAAA;AAEJ;AAKA,eAAe,0BACb,OACA,QAA6B,iCAC7B,OAC6B;AAC7B,QAAM,EAAE,MAAM,IAAI,uBAAA,IAA2B,2BAAA;AAG7C,MAAI,cAA4B;AAChC,MAAI,mBAAkC;AACtC,MAAI,mBAAmB;AAEvB,WAAS,oBAA4B;AACnC,WAAO,cAAc,KAAK,IAAA,CAAK,IAAI,EAAE,gBAAgB;AAAA,EACvD;AAEA,QAAM,SAAS,IAAI,iBAAiB,MAAM,OAAO,OAAO,KAAK;AAE7D,MAAI,UAA2B;AAAA,IAC7B,IAAI;AAAA,IACJ,MAAM;AAAA,IACN,WAAW;AAAA,IACX,OAAO,CAAA;AAAA,EAAC;AAGV,MAAI,2BAA2B;AAE/B,SAAO,UAAU,MAAM;AACrB,SAAK,iBAAiB,EAAE,QAAQ,OAAA,CAAQ;AACxC,SAAK,eAAe,EAAE,MAAM,OAAA,CAAQ;AAAA,EACtC;AAEA,SAAO,UAAU,CAAC,UAAU;AAC1B,SAAK,SAAS,EAAE,OAAO;AACvB,SAAK,iBAAiB,EAAE,QAAQ,QAAA,CAAS;AACzC,SAAK,eAAe,EAAE,MAAM,OAAA,CAAQ;AAAA,EACtC;AAEA,SAAO,oBAAoB,CAAC,aAA2B;AACrD,YAAQ,SAAS,MAAA;AAAA,MACf,KAAK;AACH,gBAAQ,MAAM,KAAK;AAAA,UACjB,MAAM;AAAA,UACN,SAAS,SAAS;AAAA,QAAA,CACnB;AACD;AAAA,MACF,KAAK,SAAS;AAEZ,cAAM,MAAM,oBAAoB,SAAS,KAAK,SAAS;AACvD,gBAAQ,MAAM,KAAK;AAAA,UACjB,MAAM;AAAA,UACN,YAAY,SAAS,KAAK;AAAA,UAC1B,WAAW;AAAA,QAAA,CACZ;AACD,YAAI,gBAAgB,YAAY;AAC9B,wBAAc;AACd,eAAK,eAAe,EAAE,MAAM,WAAA,CAAY;AAAA,QAC1C;AACA,oBAAY,KAAK,GAAG,EAAE,MAAM,CAAC,UAAU,KAAK,SAAS,EAAE,MAAA,CAAO,CAAC;AAC/D;AAAA,MACF;AAAA,MACA,KAAK;AACH,aAAK,WAAW,EAAE,UAAU,SAAS,KAAK,UAAU;AACpD;AAAA,MACF,KAAK;AACH,aAAK,SAAS;AAAA,UACZ,kBAAkB,SAAS,KAAK,sBAAsB;AAAA,UACtD,cAAc,SAAS,KAAK,oBAAoB;AAAA,UAChD,aAAa,SAAS,KAAK,mBAAmB;AAAA,QAAA,CAC/C;AACD;AAAA,MACF,KAAK;AACH,YAAI,SAAS,KAAK,YAAY,gBAAgB,YAAY;AACxD,wBAAc;AACd,eAAK,eAAe,EAAE,MAAM,WAAA,CAAY;AAAA,QAC1C;AACA,aAAK,cAAc;AAAA,UACjB,SAAS,SAAS,KAAK;AAAA,UACvB,YAAY,SAAS,KAAK;AAAA,UAC1B,MAAM;AAAA,QAAA,CACP;AACD;AAAA,MACF,KAAK;AACH,oCAA4B,SAAS,KAAK;AAC1C,aAAK,cAAc;AAAA,UACjB,SAAS,SAAS,KAAK;AAAA,UACvB,YAAY;AAAA,UACZ,MAAM;AAAA,QAAA,CACP;AACD;AAAA,MACF,KAAK;AACH,oBAAY,UAAA;AACZ,sBAAc;AACd,aAAK,eAAe,EAAE,MAAM,YAAA,CAAa;AACzC,aAAK,eAAe,EAAE,WAAW,oBAAoB,QAAW;AAChE;AAAA,MACF,KAAK;AACH,mBAAW,QAAQ,SAAS,KAAK,iBAAiB,CAAA,GAAI;AACpD,cAAI,KAAK,MAAM,KAAK,MAAM;AACxB,iBAAK,aAAa;AAAA,cAChB,YAAY,KAAK;AAAA,cACjB,OAAO,KAAK;AAAA,cACZ,UAAU,KAAK;AAAA,YAAA,CAChB;AAAA,UACH;AAAA,QACF;AACA;AAAA,MACF,KAAK;AACH,2BAAmB,kBAAA;AACnB,gBAAQ,KAAK;AACb,gBAAQ,YAAY,KAAK,IAAA;AACzB,gBAAQ,MAAM,KAAK;AAAA,UACjB,MAAM;AAAA,UACN,SAAS;AAAA,QAAA,CACV;AACD,aAAK,oBAAoB,EAAE,SAAS;AAEpC,mCAA2B;AAC3B,kBAAU;AAAA,UACR,IAAI;AAAA,UACJ,MAAM;AAAA,UACN,WAAW;AAAA,UACX,OAAO,CAAA;AAAA,QAAC;AAEV,sBAAc;AACd,aAAK,eAAe,EAAE,MAAM,YAAA,CAAa;AACzC;AAAA,MACF,KAAK;AACH,aAAK,iBAAiB,EAAE,QAAQ,YAAA,CAAa;AAC7C;AAAA,MACF,KAAK;AACH,aAAK,SAAS;AAAA,UACZ,OAAO,IAAI,MAAM,SAAS,IAAI;AAAA,QAAA,CAC/B;AACD;AAAA,IAIA;AAAA,EAEN;AAEA,QAAM,OAAO,QAAA;AAEb,QAAM,gBAAgB,IAAI,cAAc,MAAM;AAC9C,QAAM,cAAc,IAAI,YAAA;AACxB,MAAI;AACF,UAAM,YAAY,KAAA;AAClB,UAAM,cAAc,MAAA;AAAA,EACtB,SAAS,OAAO;AAGd,kBAAc,KAAA;AACd,gBAAY,QAAA;AACZ,WAAO,WAAA;AACP,UAAM;AAAA,EACR;AAEA,QAAM,aAAiC;AAAA,IACrC,MAAM,aAAa;AACjB,oBAAc,KAAA;AACd,kBAAY,QAAA;AACZ,aAAO,WAAA;AACP,oBAAc;AACd,WAAK,iBAAiB,EAAE,QAAQ,OAAA,CAAQ;AAAA,IAC1C;AAAA,IAEA,MAAM,oBAAoB;AAExB,oBAAc,kBAAA;AACd,oBAAc;AACd,WAAK,eAAe,EAAE,MAAM,YAAA,CAAa;AAAA,IAC3C;AAAA,IAEA,mBAAmB;AACjB,oBAAc,iBAAA;AACd,oBAAc;AACd,WAAK,eAAe,EAAE,MAAM,OAAA,CAAQ;AAAA,IACtC;AAAA,IAEA,SAAS,MAAc;AACrB,aAAO,gBAAgB,IAAI;AAC3B,oBAAc;AACd,WAAK,eAAe,EAAE,MAAM,WAAA,CAAY;AAAA,IAC1C;AAAA,IAEA,UAAU,WAAmB,UAAkB;AAC7C,aAAO,iBAAiB,WAAW,QAAQ;AAC3C,oBAAc;AACd,WAAK,eAAe,EAAE,MAAM,WAAA,CAAY;AAAA,IAC1C;AAAA,IAEA,eAAe,QAAgB,QAAgB;AAC7C,aAAO,iBAAiB;AAAA,QACtB;AAAA,UACE,IAAI;AAAA,UACJ,UAAU;AAAA,YACR;AAAA,UAAA;AAAA,QACF;AAAA,MACF,CACD;AAAA,IACH;AAAA,IAEA,cAAc,QAAQ;AACpB,aAAO,cAAc,MAAM,EAAE,MAAM,CAAC,UAAU,KAAK,SAAS,EAAE,MAAA,CAAO,CAAC;AACtE,WAAK,iBAAiB,EAAE,QAAQ,eAAA,CAAgB;AAAA,IAClD;AAAA,IAEA,YAAY,UAAU;AACpB,aAAO,YAAY,QAAQ;AAC3B,WAAK,iBAAiB,EAAE,QAAQ,eAAA,CAAgB;AAAA,IAClD;AAAA,IAEA,YAAY;AACV,kBAAY,UAAA;AACZ,oBAAc;AACd,WAAK,eAAe,EAAE,MAAM,YAAA,CAAa;AACzC,WAAK,eAAe,EAAE,WAAW,oBAAoB,QAAW;AAAA,IAClE;AAAA,IACA,IAAI;AAAA,IACJ,wBAA4C;AAC1C,aAAO;AAAA,QACL,IAAI,aAAa;AACf,iBAAO,cAAc;AAAA,QACvB;AAAA,QAEA,IAAI,cAAc;AAChB,iBAAO,YAAY;AAAA,QACrB;AAAA,QAEA,wBAAwB;AACtB,iBAAO,cAAc;AAAA,QACvB;AAAA,QAEA,yBAAyB;AACvB,iBAAO,YAAY;AAAA,QACrB;AAAA,QAEA,yBAAyB;AACvB,iBAAO,cAAc;AAAA,QACvB;AAAA,QAEA,0BAA0B;AACxB,iBAAO,YAAY;AAAA,QACrB;AAAA,QAEA,IAAI,kBAAkB;AACpB,iBAAO,cAAc;AAAA,QACvB;AAAA,QAEA,IAAI,mBAAmB;AACrB,iBAAO,YAAY;AAAA,QACrB;AAAA,MAAA;AAAA,IAEJ;AAAA,EAAA;AAGF,SAAO;AACT;"}
@@ -0,0 +1,98 @@
1
+ import { FunctionResponse, LiveClientMessage, LiveServerGoAway, LiveServerMessage, LiveServerSessionResumptionUpdate, LiveServerToolCall, UsageMetadata } from '@google/genai';
2
+ import { AnyClientTool, RealtimeSessionConfig, RealtimeToken } from '@tanstack/ai';
3
+ import { GeminiRealtimeModel } from './types.js';
4
+ interface LiveResponsePayloads {
5
+ text: string;
6
+ thought: string;
7
+ audio: {
8
+ audioData: string;
9
+ transcript: string;
10
+ };
11
+ setup_complete: string;
12
+ interrupted: string;
13
+ turn_complete: string;
14
+ tool_call: LiveServerToolCall;
15
+ session_resumption_update: LiveServerSessionResumptionUpdate;
16
+ go_away: LiveServerGoAway;
17
+ usage_metadata: UsageMetadata;
18
+ error: string;
19
+ input_transcription: {
20
+ text: string;
21
+ finished: boolean;
22
+ };
23
+ output_transcription: {
24
+ text: string;
25
+ finished: boolean;
26
+ };
27
+ }
28
+ export type MultimodalLiveResponseType = keyof LiveResponsePayloads;
29
+ export type LiveResponse = {
30
+ [K in MultimodalLiveResponseType]: {
31
+ type: K;
32
+ data: LiveResponsePayloads[K];
33
+ endOfTurn: boolean;
34
+ };
35
+ }[MultimodalLiveResponseType];
36
+ /**
37
+ * Parses response messages from the Gemini Live API
38
+ */
39
+ /**
40
+ * Parses ALL response types from a single server message.
41
+ * The server can now bundle multiple fields (e.g. audio + transcription)
42
+ * in the same message. Returns an array of response objects.
43
+ */
44
+ export declare function parseResponseMessages(data: LiveServerMessage): Array<LiveResponse>;
45
+ export declare class GeminiLiveClient {
46
+ private token;
47
+ private model;
48
+ private readonly responseModalities;
49
+ private systemInstructions;
50
+ private googleGrounding;
51
+ private voiceName;
52
+ private temperature;
53
+ private inputAudioTranscription;
54
+ private outputAudioTranscription;
55
+ private contextWindowCompression;
56
+ private proactiveAudio;
57
+ private enableAffectiveDialog;
58
+ private thinkingConfig;
59
+ private speechLanguageCode;
60
+ private maxOutputTokens;
61
+ private functionDeclarations;
62
+ private readonly automaticActivityDetection;
63
+ private readonly activityHandling;
64
+ private webSocket;
65
+ private lastResumptionUpdate;
66
+ private setupComplete;
67
+ private connected;
68
+ onReceiveResponse: (response: LiveResponse) => void;
69
+ onOpen: () => void;
70
+ onClose: () => void;
71
+ onError: (error: Error) => void;
72
+ constructor(token: string, model: GeminiRealtimeModel, tools?: ReadonlyArray<AnyClientTool>);
73
+ get isConnected(): boolean;
74
+ get isSetupComplete(): boolean;
75
+ /**
76
+ * Connection management
77
+ */
78
+ connect(): Promise<void>;
79
+ disconnect(): void;
80
+ /**
81
+ * Session management
82
+ */
83
+ sendInitialSetupMessage(resume?: boolean): void;
84
+ restartSession(resume?: boolean): Promise<void>;
85
+ updateToken(token: RealtimeToken): void;
86
+ updateSession(config: Partial<RealtimeSessionConfig>): Promise<void>;
87
+ /**
88
+ * Message transmission & receiving
89
+ */
90
+ sendMessage(message: LiveClientMessage): void;
91
+ onReceiveMessage(messageEvent: MessageEvent): Promise<void>;
92
+ sendRealtimeInputMessage(data: string, mimeType: string): void;
93
+ sendAudioMessage(base64PCM: string): void;
94
+ sendImageMessage(base64: string, mimeType?: string): void;
95
+ sendTextMessage(text: string): void;
96
+ sendToolResponse(functionResponses: Array<FunctionResponse>): void;
97
+ }
98
+ export {};