@tanstack/ai-gemini 0.9.1 → 0.10.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -74,7 +74,7 @@ export class GeminiImageAdapter<
74
74
  private client: GoogleGenAI
75
75
 
76
76
  constructor(config: GeminiImageConfig, model: TModel) {
77
- super({}, model)
77
+ super(model, config)
78
78
  this.client = createGeminiClient(config)
79
79
  }
80
80
 
@@ -138,8 +138,24 @@ export class GeminiImageAdapter<
138
138
  ? `${prompt} Generate ${numberOfImages} distinct images.`
139
139
  : prompt
140
140
 
141
+ // GeminiImageProviderOptions is Imagen-shaped — most fields
142
+ // (personGeneration, safetyFilterLevel, addWatermark, outputMimeType,
143
+ // outputCompressionQuality, guidanceScale, enhancePrompt,
144
+ // includeSafetyAttributes, includeRaiReason, outputGcsUri, labels,
145
+ // negativePrompt, language) are only valid on GenerateImagesConfig and
146
+ // would be rejected by the Gemini-native generateContent path. Pick only
147
+ // the fields that are valid on GenerateContentConfig instead of spreading
148
+ // the whole options object.
149
+ const nativeConfig: GenerateContentConfig = {}
150
+ if (modelOptions?.seed !== undefined) {
151
+ nativeConfig.seed = modelOptions.seed
152
+ }
153
+
141
154
  const config: GenerateContentConfig = {
142
- // Include TEXT so the model can interleave descriptions between images
155
+ ...nativeConfig,
156
+ // Include TEXT so the model can interleave descriptions between images.
157
+ // IMPORTANT: responseModalities is a protected default — set it AFTER
158
+ // nativeConfig so nothing can silently disable image output.
143
159
  responseModalities: ['TEXT', 'IMAGE'],
144
160
  ...(parsedSize && {
145
161
  imageConfig: {
@@ -151,7 +167,6 @@ export class GeminiImageAdapter<
151
167
  }),
152
168
  },
153
169
  }),
154
- ...modelOptions,
155
170
  }
156
171
 
157
172
  const response = await this.client.models.generateContent({
@@ -168,6 +183,7 @@ export class GeminiImageAdapter<
168
183
  response: GenerateContentResponse,
169
184
  ): ImageGenerationResult {
170
185
  const images: Array<GeneratedImage> = []
186
+ const textParts: Array<string> = []
171
187
  const parts = response.candidates?.[0]?.content?.parts ?? []
172
188
 
173
189
  for (const part of parts) {
@@ -177,9 +193,23 @@ export class GeminiImageAdapter<
177
193
  part.inlineData.data.length > 0
178
194
  ) {
179
195
  images.push({ b64Json: part.inlineData.data })
196
+ } else if (typeof part.text === 'string' && part.text.length > 0) {
197
+ textParts.push(part.text)
180
198
  }
181
199
  }
182
200
 
201
+ // If the model returned only text parts (for example a safety refusal
202
+ // or a "can't do that" message), surface the text instead of silently
203
+ // resolving to an empty images array — otherwise callers can't tell a
204
+ // generation failure apart from a genuine empty response.
205
+ if (images.length === 0) {
206
+ const reason =
207
+ textParts.length > 0
208
+ ? `: ${textParts.join(' ').trim()}`
209
+ : ' (no inline image or text parts were returned).'
210
+ throw new Error(`Gemini ${model} returned no images${reason}`)
211
+ }
212
+
183
213
  return {
184
214
  id: generateId(this.name),
185
215
  model,
@@ -205,12 +235,43 @@ export class GeminiImageAdapter<
205
235
  model: string,
206
236
  response: GenerateImagesResponse,
207
237
  ): ImageGenerationResult {
208
- const images: Array<GeneratedImage> = (response.generatedImages ?? []).map(
209
- (item) => ({
210
- b64Json: item.image?.imageBytes,
211
- revisedPrompt: item.enhancedPrompt,
212
- }),
213
- )
238
+ const entries = response.generatedImages ?? []
239
+ const images: Array<GeneratedImage> = []
240
+ const filterReasons: Array<string> = []
241
+
242
+ for (const item of entries) {
243
+ const b64Json = item.image?.imageBytes
244
+ if (b64Json) {
245
+ images.push({ b64Json, revisedPrompt: item.enhancedPrompt })
246
+ continue
247
+ }
248
+ // Imagen can drop individual entries with a raiFilteredReason when
249
+ // Responsible-AI filters fire. Preserve the reason so callers can
250
+ // surface it instead of silently getting back fewer images.
251
+ const reason = (item as { raiFilteredReason?: string }).raiFilteredReason
252
+ if (reason) {
253
+ filterReasons.push(reason)
254
+ }
255
+ }
256
+
257
+ // Every entry was filtered — no usable images to return. Throw rather
258
+ // than resolve to an empty array so the caller is forced to handle the
259
+ // failure mode explicitly.
260
+ if (entries.length > 0 && images.length === 0) {
261
+ const joined = filterReasons.length > 0 ? filterReasons.join('; ') : ''
262
+ throw new Error(
263
+ `Imagen ${model} returned no images: all ${entries.length} generated image(s) were filtered by Responsible-AI${joined ? ` (${joined})` : ''}.`,
264
+ )
265
+ }
266
+
267
+ // Partial filter: surface via console.warn since ImageGenerationResult
268
+ // has no warnings field. Callers that care can still inspect the count
269
+ // mismatch between requested and returned images.
270
+ if (filterReasons.length > 0 && typeof console !== 'undefined') {
271
+ console.warn(
272
+ `[gemini-image] ${filterReasons.length} of ${entries.length} images from ${model} were filtered by Responsible-AI: ${filterReasons.join('; ')}`,
273
+ )
274
+ }
214
275
 
215
276
  return {
216
277
  id: generateId(this.name),
@@ -4,21 +4,40 @@ import {
4
4
  generateId,
5
5
  getGeminiApiKeyFromEnv,
6
6
  } from '../utils'
7
+ import { GEMINI_TTS_VOICES } from '../model-meta'
7
8
  import type { GEMINI_TTS_MODELS, GeminiTTSVoice } from '../model-meta'
8
9
  import type { TTSOptions, TTSResult } from '@tanstack/ai'
9
- import type { GoogleGenAI } from '@google/genai'
10
+ import type { GoogleGenAI, SpeechConfig } from '@google/genai'
10
11
  import type { GeminiClientConfig } from '../utils'
11
12
 
13
+ /**
14
+ * Configuration for a single speaker in a multi-speaker dialogue.
15
+ * Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.
16
+ */
17
+ export interface GeminiSpeakerVoiceConfig {
18
+ /** A name used in the prompt to refer to this speaker */
19
+ speaker: string
20
+ /** Voice configuration for this speaker */
21
+ voiceConfig: {
22
+ prebuiltVoiceConfig: {
23
+ voiceName: GeminiTTSVoice
24
+ }
25
+ }
26
+ }
27
+
12
28
  /**
13
29
  * Provider-specific options for Gemini TTS
14
30
  *
15
31
  * @experimental Gemini TTS is an experimental feature.
16
32
  * @see https://ai.google.dev/gemini-api/docs/speech-generation
33
+ * @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview
17
34
  */
18
35
  export interface GeminiTTSProviderOptions {
19
36
  /**
20
- * Voice configuration for TTS.
37
+ * Voice configuration for single-speaker TTS.
21
38
  * Choose from 30 available voices with different characteristics.
39
+ *
40
+ * Use `multiSpeakerVoiceConfig` instead for dialogues.
22
41
  */
23
42
  voiceConfig?: {
24
43
  prebuiltVoiceConfig?: {
@@ -30,11 +49,31 @@ export interface GeminiTTSProviderOptions {
30
49
  }
31
50
  }
32
51
 
52
+ /**
53
+ * Multi-speaker voice configuration (up to 2 speakers).
54
+ * Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.
55
+ *
56
+ * Each speaker's lines in the prompt are prefixed with the name defined
57
+ * here, e.g.:
58
+ *
59
+ * ```text
60
+ * Joe: Hey, how's it going?
61
+ * Jane: Not bad, you?
62
+ * ```
63
+ */
64
+ multiSpeakerVoiceConfig?: {
65
+ speakerVoiceConfigs: Array<GeminiSpeakerVoiceConfig>
66
+ }
67
+
33
68
  /**
34
69
  * System instruction for controlling speech style.
35
70
  * Use natural language to describe the desired speaking style,
36
71
  * pace, tone, accent, or other characteristics.
37
72
  *
73
+ * With Gemini 3.1 Flash TTS, you can also use inline audio tags like
74
+ * `[whispering]`, `[laughs]`, `[excited]` directly in the input text
75
+ * to control delivery.
76
+ *
38
77
  * @example "Speak slowly and calmly, as if telling a bedtime story"
39
78
  * @example "Use an upbeat, enthusiastic tone with moderate pace"
40
79
  * @example "Speak with a British accent"
@@ -43,8 +82,8 @@ export interface GeminiTTSProviderOptions {
43
82
 
44
83
  /**
45
84
  * Language code hint for the speech synthesis.
46
- * Gemini TTS supports 24 languages and can auto-detect,
47
- * but you can provide a hint for better results.
85
+ * Gemini 3.1 Flash TTS supports 70+ languages with auto-detection;
86
+ * the 2.5 TTS models support 24 languages.
48
87
  *
49
88
  * @example "en-US" for American English
50
89
  * @example "es-ES" for Spanish (Spain)
@@ -85,7 +124,7 @@ export class GeminiTTSAdapter<
85
124
  private client: GoogleGenAI
86
125
 
87
126
  constructor(config: GeminiTTSConfig, model: TModel) {
88
- super(config, model)
127
+ super(model, config)
89
128
  this.client = createGeminiClient(config)
90
129
  }
91
130
 
@@ -98,18 +137,55 @@ export class GeminiTTSAdapter<
98
137
  async generateSpeech(
99
138
  options: TTSOptions<GeminiTTSProviderOptions>,
100
139
  ): Promise<TTSResult> {
101
- const { logger } = options
102
- const { model, text, modelOptions } = options
140
+ const { model, text, modelOptions, voice, logger } = options
103
141
 
104
142
  logger.request(`activity=generateSpeech provider=gemini model=${model}`, {
105
143
  provider: 'gemini',
106
144
  model,
107
145
  })
108
146
 
109
- const voiceConfig = modelOptions?.voiceConfig || {
110
- prebuiltVoiceConfig: {
111
- voiceName: 'Kore',
112
- },
147
+ const speechConfig: SpeechConfig = {}
148
+
149
+ if (modelOptions?.multiSpeakerVoiceConfig) {
150
+ // Validate multi-speaker config: 1 or 2 speakers allowed.
151
+ const speakerConfigs =
152
+ modelOptions.multiSpeakerVoiceConfig.speakerVoiceConfigs
153
+ if (
154
+ !Array.isArray(speakerConfigs) ||
155
+ speakerConfigs.length < 1 ||
156
+ speakerConfigs.length > 2
157
+ ) {
158
+ throw new Error(
159
+ `Gemini TTS multiSpeakerVoiceConfig.speakerVoiceConfigs must contain 1 or 2 speakers; received ${Array.isArray(speakerConfigs) ? speakerConfigs.length : 'non-array'}.`,
160
+ )
161
+ }
162
+ speechConfig.multiSpeakerVoiceConfig =
163
+ modelOptions.multiSpeakerVoiceConfig
164
+ } else {
165
+ // Honor the standard TTSOptions.voice (used by every other TTS adapter)
166
+ // as a fallback for the prebuilt voice name. If an explicit
167
+ // modelOptions.voiceConfig is supplied its values win — but we still
168
+ // fall back to `voice` / 'Kore' if the supplied voiceConfig is missing
169
+ // prebuiltVoiceConfig.voiceName.
170
+ if (
171
+ voice !== undefined &&
172
+ !(GEMINI_TTS_VOICES as ReadonlyArray<string>).includes(voice)
173
+ ) {
174
+ throw new Error(
175
+ `Invalid Gemini TTS voice "${voice}". Valid voices are: ${GEMINI_TTS_VOICES.join(', ')}.`,
176
+ )
177
+ }
178
+ const defaultVoiceName = (voice as GeminiTTSVoice | undefined) ?? 'Kore'
179
+ const supplied = modelOptions?.voiceConfig
180
+ const resolvedVoiceName =
181
+ supplied?.prebuiltVoiceConfig?.voiceName ?? defaultVoiceName
182
+ speechConfig.voiceConfig = {
183
+ prebuiltVoiceConfig: { voiceName: resolvedVoiceName },
184
+ }
185
+ }
186
+
187
+ if (modelOptions?.languageCode) {
188
+ speechConfig.languageCode = modelOptions.languageCode
113
189
  }
114
190
 
115
191
  try {
@@ -123,16 +199,13 @@ export class GeminiTTSAdapter<
123
199
  ],
124
200
  config: {
125
201
  responseModalities: ['AUDIO'],
126
- speechConfig: {
127
- voiceConfig,
128
- ...(modelOptions?.languageCode && {
129
- languageCode: modelOptions.languageCode,
130
- }),
131
- },
202
+ speechConfig,
203
+ // systemInstruction belongs inside `config` per the @google/genai
204
+ // contract — matches sibling Gemini adapters (summarize, text).
205
+ ...(modelOptions?.systemInstruction && {
206
+ systemInstruction: modelOptions.systemInstruction,
207
+ }),
132
208
  },
133
- ...(modelOptions?.systemInstruction && {
134
- systemInstruction: modelOptions.systemInstruction,
135
- }),
136
209
  })
137
210
 
138
211
  // Extract audio data from response
@@ -153,8 +226,33 @@ export class GeminiTTSAdapter<
153
226
  }
154
227
 
155
228
  const audioBase64 = audioPart.inlineData.data
156
- const mimeType = audioPart.inlineData.mimeType || 'audio/wav'
157
- const format = mimeType.split('/')[1] || 'wav'
229
+ // mime is guaranteed by the `startsWith('audio/')` find predicate above.
230
+ const mimeType = audioPart.inlineData.mimeType as string
231
+
232
+ // Gemini TTS models return raw 16-bit LE PCM with a mime type like
233
+ // `audio/L16;codec=pcm;rate=24000`. That isn't playable in an <audio>
234
+ // element and the bare string isn't a usable file extension, so we
235
+ // prepend a RIFF/WAV header here and normalize the result to audio/wav.
236
+ const pcm = parsePcmMimeType(mimeType)
237
+ if (pcm) {
238
+ const wavBase64 = wrapPcmBase64AsWav(
239
+ audioBase64,
240
+ pcm.sampleRate,
241
+ pcm.channels,
242
+ pcm.bitsPerSample,
243
+ )
244
+ return {
245
+ id: generateId(this.name),
246
+ model,
247
+ audio: wavBase64,
248
+ format: 'wav',
249
+ contentType: 'audio/wav',
250
+ }
251
+ }
252
+
253
+ // Strip any mime parameters (e.g. `audio/ogg;codec=opus`) before pulling
254
+ // the subtype out as the file format.
255
+ const format = mimeType.split(';')[0]!.split('/')[1] || 'wav'
158
256
 
159
257
  return {
160
258
  id: generateId(this.name),
@@ -173,6 +271,105 @@ export class GeminiTTSAdapter<
173
271
  }
174
272
  }
175
273
 
274
+ function parsePcmMimeType(
275
+ mimeType: string,
276
+ ): { sampleRate: number; channels: number; bitsPerSample: number } | undefined {
277
+ const normalized = mimeType.toLowerCase()
278
+ const subtype = normalized.split(';')[0]!.split('/')[1] ?? ''
279
+ // Exclude containerized wav (e.g. `audio/wav;codec=pcm`) — those already
280
+ // carry a RIFF header and must not be re-wrapped.
281
+ if (subtype.includes('wav')) return undefined
282
+
283
+ // Accept the variants Gemini and other providers actually emit:
284
+ // - audio/L16;codec=pcm;rate=24000 (IANA PCM with bit depth in the type)
285
+ // - audio/L24 and friends
286
+ // - audio/pcm and audio/x-pcm
287
+ // - anything else that explicitly tags codec=pcm and isn't wav-containered
288
+ const bitDepthMatch = /^audio\/l(\d+)/.exec(normalized)
289
+ const isPcm =
290
+ bitDepthMatch !== null ||
291
+ normalized.startsWith('audio/pcm') ||
292
+ normalized.startsWith('audio/x-pcm') ||
293
+ normalized.includes('codec=pcm')
294
+ if (!isPcm) return undefined
295
+
296
+ const rateMatch = /rate=(\d+)/.exec(normalized)
297
+ const channelsMatch = /channels=(\d+)/.exec(normalized)
298
+ // Default to 16-bit when the mime type doesn't specify — matches Gemini's
299
+ // audio/L16;codec=pcm;rate=24000 response.
300
+ const bitsPerSample = bitDepthMatch ? Number(bitDepthMatch[1]) : 16
301
+ return {
302
+ sampleRate: rateMatch ? Number(rateMatch[1]) : 24000,
303
+ channels: channelsMatch ? Number(channelsMatch[1]) : 1,
304
+ bitsPerSample,
305
+ }
306
+ }
307
+
308
+ function wrapPcmBase64AsWav(
309
+ pcmBase64: string,
310
+ sampleRate: number,
311
+ channels = 1,
312
+ bitsPerSample = 16,
313
+ ): string {
314
+ // The WAV writer below emits a 16-bit PCM fmt chunk. If the source claims a
315
+ // different bit depth we'd be lying about the payload, so bail out loudly
316
+ // rather than producing a corrupt file.
317
+ if (bitsPerSample !== 16) {
318
+ throw new Error(
319
+ `Unsupported PCM bit depth ${bitsPerSample}: only 16-bit PCM can be wrapped as WAV.`,
320
+ )
321
+ }
322
+
323
+ const pcmBytes =
324
+ typeof Buffer !== 'undefined'
325
+ ? new Uint8Array(Buffer.from(pcmBase64, 'base64'))
326
+ : decodeBase64(pcmBase64)
327
+
328
+ const byteRate = (sampleRate * channels * bitsPerSample) / 8
329
+ const blockAlign = (channels * bitsPerSample) / 8
330
+ const dataSize = pcmBytes.byteLength
331
+ const buffer = new ArrayBuffer(44 + dataSize)
332
+ const view = new DataView(buffer)
333
+
334
+ writeAscii(view, 0, 'RIFF')
335
+ view.setUint32(4, 36 + dataSize, true)
336
+ writeAscii(view, 8, 'WAVE')
337
+ writeAscii(view, 12, 'fmt ')
338
+ view.setUint32(16, 16, true)
339
+ view.setUint16(20, 1, true)
340
+ view.setUint16(22, channels, true)
341
+ view.setUint32(24, sampleRate, true)
342
+ view.setUint32(28, byteRate, true)
343
+ view.setUint16(32, blockAlign, true)
344
+ view.setUint16(34, bitsPerSample, true)
345
+ writeAscii(view, 36, 'data')
346
+ view.setUint32(40, dataSize, true)
347
+ new Uint8Array(buffer, 44).set(pcmBytes)
348
+
349
+ if (typeof Buffer !== 'undefined') {
350
+ return Buffer.from(buffer).toString('base64')
351
+ }
352
+ let binary = ''
353
+ const bytes = new Uint8Array(buffer)
354
+ for (let i = 0; i < bytes.byteLength; i += 1) {
355
+ binary += String.fromCharCode(bytes[i]!)
356
+ }
357
+ return btoa(binary)
358
+ }
359
+
360
+ function decodeBase64(b64: string): Uint8Array {
361
+ const binary = atob(b64)
362
+ const out = new Uint8Array(binary.length)
363
+ for (let i = 0; i < binary.length; i += 1) out[i] = binary.charCodeAt(i)
364
+ return out
365
+ }
366
+
367
+ function writeAscii(view: DataView, offset: number, text: string): void {
368
+ for (let i = 0; i < text.length; i += 1) {
369
+ view.setUint8(offset + i, text.charCodeAt(i))
370
+ }
371
+ }
372
+
176
373
  /**
177
374
  * Creates a Gemini TTS adapter with explicit API key.
178
375
  * Type resolution happens here at the call site.
@@ -199,7 +396,9 @@ export function createGeminiSpeech<TModel extends GeminiTTSModel>(
199
396
  apiKey: string,
200
397
  config?: Omit<GeminiTTSConfig, 'apiKey'>,
201
398
  ): GeminiTTSAdapter<TModel> {
202
- return new GeminiTTSAdapter({ apiKey, ...config }, model)
399
+ // Put apiKey LAST so caller-supplied config can't silently override the
400
+ // explicit argument.
401
+ return new GeminiTTSAdapter({ ...config, apiKey }, model)
203
402
  }
204
403
 
205
404
  /**
@@ -244,8 +244,28 @@ export function validateImageSize(
244
244
  }
245
245
 
246
246
  /**
247
- * Validates the number of images requested
248
- * Imagen models support 1-8 images per request (varies by model)
247
+ * Per-model caps on images per request.
248
+ * Imagen 3 and the Imagen 4 family all support up to 4 images per request
249
+ * via the Gemini API (the rumored 8-image tier is Vertex-only and isn't
250
+ * reachable through @google/genai today). Unknown models fall through to
251
+ * the shared cap defined below.
252
+ *
253
+ * @see https://ai.google.dev/gemini-api/docs/imagen
254
+ */
255
+ const IMAGEN_MAX_IMAGES_BY_MODEL: Record<string, number> = {
256
+ 'imagen-3.0-generate-002': 4,
257
+ 'imagen-4.0-generate-001': 4,
258
+ 'imagen-4.0-ultra-generate-001': 4,
259
+ 'imagen-4.0-fast-generate-001': 4,
260
+ }
261
+
262
+ const DEFAULT_IMAGEN_MAX_IMAGES = 4
263
+
264
+ /**
265
+ * Validates the number of images requested against the model's known cap.
266
+ * Uses a per-model table where available and falls back to the shared
267
+ * default otherwise — no more "some support up to 8" comments that don't
268
+ * match the error message.
249
269
  */
250
270
  export function validateNumberOfImages(
251
271
  model: string,
@@ -253,8 +273,8 @@ export function validateNumberOfImages(
253
273
  ): void {
254
274
  if (numberOfImages === undefined) return
255
275
 
256
- // Most Imagen models support 1-4 images, some support up to 8
257
- const maxImages = 4
276
+ const maxImages =
277
+ IMAGEN_MAX_IMAGES_BY_MODEL[model] ?? DEFAULT_IMAGEN_MAX_IMAGES
258
278
  if (numberOfImages < 1 || numberOfImages > maxImages) {
259
279
  throw new Error(
260
280
  `Invalid numberOfImages "${numberOfImages}" for model "${model}". ` +
package/src/index.ts CHANGED
@@ -51,12 +51,26 @@ export {
51
51
  type GeminiTTSProviderOptions,
52
52
  } from './adapters/tts'
53
53
 
54
+ // Audio / Lyria music generation adapter (experimental)
55
+ /**
56
+ * @experimental Gemini Lyria music generation is an experimental feature and may change.
57
+ */
58
+ export {
59
+ GeminiAudioAdapter,
60
+ createGeminiAudio,
61
+ geminiAudio,
62
+ type GeminiAudioConfig,
63
+ type GeminiAudioModel,
64
+ type GeminiAudioProviderOptions,
65
+ } from './adapters/audio'
66
+
54
67
  // Re-export models from model-meta for convenience
55
68
  export { GEMINI_MODELS } from './model-meta'
56
69
  export { GEMINI_MODELS as GeminiTextModels } from './model-meta'
57
70
  export { GEMINI_IMAGE_MODELS as GeminiImageModels } from './model-meta'
58
71
  export { GEMINI_TTS_MODELS as GeminiTTSModels } from './model-meta'
59
72
  export { GEMINI_TTS_VOICES as GeminiTTSVoices } from './model-meta'
73
+ export { GEMINI_AUDIO_MODELS as GeminiAudioModels } from './model-meta'
60
74
  export type { GeminiModels as GeminiTextModel } from './model-meta'
61
75
  export type { GeminiImageModels as GeminiImageModel } from './model-meta'
62
76
  export type { GeminiTTSVoice } from './model-meta'
package/src/model-meta.ts CHANGED
@@ -295,7 +295,6 @@ const GEMINI_2_5_PRO_TTS = {
295
295
  input: ['text'],
296
296
  output: ['audio'],
297
297
  capabilities: ['audio_generation'],
298
- tools: ['file_search'],
299
298
  },
300
299
  pricing: {
301
300
  input: {
@@ -455,7 +454,6 @@ const GEMINI_2_5_FLASH_TTS = {
455
454
  input: ['text'],
456
455
  output: ['audio'],
457
456
  capabilities: ['audio_generation', 'batch_api'],
458
- tools: ['file_search'],
459
457
  },
460
458
  pricing: {
461
459
  input: {
@@ -472,6 +470,82 @@ const GEMINI_2_5_FLASH_TTS = {
472
470
  GeminiCachedContentOptions
473
471
  >
474
472
 
473
+ /**
474
+ * Gemini 3.1 Flash TTS Preview - latest expressive TTS model with
475
+ * 200+ audio tags, 70+ languages, and multi-speaker dialogue support.
476
+ * @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview
477
+ */
478
+ const GEMINI_3_1_FLASH_TTS = {
479
+ name: 'gemini-3.1-flash-tts-preview',
480
+ max_input_tokens: 32_768,
481
+ max_output_tokens: 16_384,
482
+ knowledge_cutoff: '2025-05-01',
483
+ supports: {
484
+ input: ['text'],
485
+ output: ['audio'],
486
+ capabilities: ['audio_generation', 'batch_api'],
487
+ },
488
+ pricing: {
489
+ input: {
490
+ normal: 0.5,
491
+ },
492
+ output: {
493
+ normal: 10,
494
+ },
495
+ },
496
+ } as const satisfies ModelMeta<
497
+ GeminiToolConfigOptions &
498
+ GeminiSafetyOptions &
499
+ GeminiCommonConfigOptions &
500
+ GeminiCachedContentOptions
501
+ >
502
+
503
+ /**
504
+ * Lyria 3 Pro Preview — Google's flagship music generation model.
505
+ * Generates full-length songs with multiple verses, choruses, and bridges.
506
+ * Outputs MP3 or WAV at 48 kHz stereo.
507
+ * @see https://ai.google.dev/gemini-api/docs/models/lyria-3-pro-preview
508
+ */
509
+ const LYRIA_3_PRO = {
510
+ name: 'lyria-3-pro-preview',
511
+ max_input_tokens: 131_072,
512
+ supports: {
513
+ input: ['text', 'image'],
514
+ output: ['audio'],
515
+ capabilities: ['audio_generation'],
516
+ },
517
+ pricing: {
518
+ input: {
519
+ normal: 0,
520
+ },
521
+ output: {
522
+ normal: 0,
523
+ },
524
+ },
525
+ } as const satisfies ModelMeta
526
+
527
+ /**
528
+ * Lyria 3 Clip Preview — 30-second music clips in MP3 format.
529
+ * @see https://ai.google.dev/gemini-api/docs/music-generation
530
+ */
531
+ const LYRIA_3_CLIP = {
532
+ name: 'lyria-3-clip-preview',
533
+ max_input_tokens: 131_072,
534
+ supports: {
535
+ input: ['text', 'image'],
536
+ output: ['audio'],
537
+ capabilities: ['audio_generation'],
538
+ },
539
+ pricing: {
540
+ input: {
541
+ normal: 0,
542
+ },
543
+ output: {
544
+ normal: 0,
545
+ },
546
+ },
547
+ } as const satisfies ModelMeta
548
+
475
549
  const GEMINI_2_5_FLASH_LITE = {
476
550
  name: 'gemini-2.5-flash-lite',
477
551
  max_input_tokens: 1_048_576,
@@ -580,7 +654,7 @@ const GEMINI_2_FLASH_IMAGE = {
580
654
  knowledge_cutoff: '2024-08-01',
581
655
  supports: {
582
656
  input: ['text', 'image', 'audio', 'video'],
583
- output: ['text'],
657
+ output: ['text', 'image'],
584
658
  capabilities: ['batch_api', 'caching', 'structured_output'],
585
659
  tools: [],
586
660
  },
@@ -930,10 +1004,20 @@ export const GEMINI_IMAGE_MODELS = [
930
1004
  * @experimental Gemini TTS is an experimental feature and may change.
931
1005
  */
932
1006
  export const GEMINI_TTS_MODELS = [
1007
+ GEMINI_3_1_FLASH_TTS.name,
933
1008
  GEMINI_2_5_FLASH_TTS.name,
934
1009
  GEMINI_2_5_PRO_TTS.name,
935
1010
  ] as const
936
1011
 
1012
+ /**
1013
+ * Audio generation models (Lyria music generation).
1014
+ * @experimental Lyria music generation is an experimental feature and may change.
1015
+ */
1016
+ export const GEMINI_AUDIO_MODELS = [
1017
+ LYRIA_3_PRO.name,
1018
+ LYRIA_3_CLIP.name,
1019
+ ] as const
1020
+
937
1021
  /**
938
1022
  * Available voice names for Gemini TTS
939
1023
  * @see https://ai.google.dev/gemini-api/docs/speech-generation