@tanstack/ai-gemini 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +92 -0
- package/dist/esm/adapters/audio.js +61 -0
- package/dist/esm/adapters/audio.js.map +1 -0
- package/dist/esm/adapters/image.js +70 -23
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/summarize.js +98 -70
- package/dist/esm/adapters/summarize.js.map +1 -1
- package/dist/esm/adapters/text.js +21 -2
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/tts.d.ts +39 -3
- package/dist/esm/adapters/tts.js +153 -44
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/image/image-provider-options.d.ts +4 -2
- package/dist/esm/image/image-provider-options.js +8 -1
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +5 -0
- package/dist/esm/index.js +6 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +6 -1
- package/dist/esm/model-meta.js +15 -0
- package/dist/esm/model-meta.js.map +1 -1
- package/package.json +3 -3
- package/src/adapters/audio.ts +185 -0
- package/src/adapters/image.ts +101 -24
- package/src/adapters/summarize.ts +111 -80
- package/src/adapters/text.ts +22 -1
- package/src/adapters/tts.ts +264 -51
- package/src/image/image-provider-options.ts +24 -4
- package/src/index.ts +14 -0
- package/src/model-meta.ts +87 -3
package/src/adapters/tts.ts
CHANGED
|
@@ -4,21 +4,40 @@ import {
|
|
|
4
4
|
generateId,
|
|
5
5
|
getGeminiApiKeyFromEnv,
|
|
6
6
|
} from '../utils'
|
|
7
|
+
import { GEMINI_TTS_VOICES } from '../model-meta'
|
|
7
8
|
import type { GEMINI_TTS_MODELS, GeminiTTSVoice } from '../model-meta'
|
|
8
9
|
import type { TTSOptions, TTSResult } from '@tanstack/ai'
|
|
9
|
-
import type { GoogleGenAI } from '@google/genai'
|
|
10
|
+
import type { GoogleGenAI, SpeechConfig } from '@google/genai'
|
|
10
11
|
import type { GeminiClientConfig } from '../utils'
|
|
11
12
|
|
|
13
|
+
/**
|
|
14
|
+
* Configuration for a single speaker in a multi-speaker dialogue.
|
|
15
|
+
* Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.
|
|
16
|
+
*/
|
|
17
|
+
export interface GeminiSpeakerVoiceConfig {
|
|
18
|
+
/** A name used in the prompt to refer to this speaker */
|
|
19
|
+
speaker: string
|
|
20
|
+
/** Voice configuration for this speaker */
|
|
21
|
+
voiceConfig: {
|
|
22
|
+
prebuiltVoiceConfig: {
|
|
23
|
+
voiceName: GeminiTTSVoice
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
|
|
12
28
|
/**
|
|
13
29
|
* Provider-specific options for Gemini TTS
|
|
14
30
|
*
|
|
15
31
|
* @experimental Gemini TTS is an experimental feature.
|
|
16
32
|
* @see https://ai.google.dev/gemini-api/docs/speech-generation
|
|
33
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview
|
|
17
34
|
*/
|
|
18
35
|
export interface GeminiTTSProviderOptions {
|
|
19
36
|
/**
|
|
20
|
-
* Voice configuration for TTS.
|
|
37
|
+
* Voice configuration for single-speaker TTS.
|
|
21
38
|
* Choose from 30 available voices with different characteristics.
|
|
39
|
+
*
|
|
40
|
+
* Use `multiSpeakerVoiceConfig` instead for dialogues.
|
|
22
41
|
*/
|
|
23
42
|
voiceConfig?: {
|
|
24
43
|
prebuiltVoiceConfig?: {
|
|
@@ -30,11 +49,31 @@ export interface GeminiTTSProviderOptions {
|
|
|
30
49
|
}
|
|
31
50
|
}
|
|
32
51
|
|
|
52
|
+
/**
|
|
53
|
+
* Multi-speaker voice configuration (up to 2 speakers).
|
|
54
|
+
* Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.
|
|
55
|
+
*
|
|
56
|
+
* Each speaker's lines in the prompt are prefixed with the name defined
|
|
57
|
+
* here, e.g.:
|
|
58
|
+
*
|
|
59
|
+
* ```text
|
|
60
|
+
* Joe: Hey, how's it going?
|
|
61
|
+
* Jane: Not bad, you?
|
|
62
|
+
* ```
|
|
63
|
+
*/
|
|
64
|
+
multiSpeakerVoiceConfig?: {
|
|
65
|
+
speakerVoiceConfigs: Array<GeminiSpeakerVoiceConfig>
|
|
66
|
+
}
|
|
67
|
+
|
|
33
68
|
/**
|
|
34
69
|
* System instruction for controlling speech style.
|
|
35
70
|
* Use natural language to describe the desired speaking style,
|
|
36
71
|
* pace, tone, accent, or other characteristics.
|
|
37
72
|
*
|
|
73
|
+
* With Gemini 3.1 Flash TTS, you can also use inline audio tags like
|
|
74
|
+
* `[whispering]`, `[laughs]`, `[excited]` directly in the input text
|
|
75
|
+
* to control delivery.
|
|
76
|
+
*
|
|
38
77
|
* @example "Speak slowly and calmly, as if telling a bedtime story"
|
|
39
78
|
* @example "Use an upbeat, enthusiastic tone with moderate pace"
|
|
40
79
|
* @example "Speak with a British accent"
|
|
@@ -43,8 +82,8 @@ export interface GeminiTTSProviderOptions {
|
|
|
43
82
|
|
|
44
83
|
/**
|
|
45
84
|
* Language code hint for the speech synthesis.
|
|
46
|
-
* Gemini TTS supports
|
|
47
|
-
*
|
|
85
|
+
* Gemini 3.1 Flash TTS supports 70+ languages with auto-detection;
|
|
86
|
+
* the 2.5 TTS models support 24 languages.
|
|
48
87
|
*
|
|
49
88
|
* @example "en-US" for American English
|
|
50
89
|
* @example "es-ES" for Spanish (Spain)
|
|
@@ -85,7 +124,7 @@ export class GeminiTTSAdapter<
|
|
|
85
124
|
private client: GoogleGenAI
|
|
86
125
|
|
|
87
126
|
constructor(config: GeminiTTSConfig, model: TModel) {
|
|
88
|
-
super(
|
|
127
|
+
super(model, config)
|
|
89
128
|
this.client = createGeminiClient(config)
|
|
90
129
|
}
|
|
91
130
|
|
|
@@ -98,64 +137,236 @@ export class GeminiTTSAdapter<
|
|
|
98
137
|
async generateSpeech(
|
|
99
138
|
options: TTSOptions<GeminiTTSProviderOptions>,
|
|
100
139
|
): Promise<TTSResult> {
|
|
101
|
-
const { model, text, modelOptions } = options
|
|
140
|
+
const { model, text, modelOptions, voice, logger } = options
|
|
141
|
+
|
|
142
|
+
logger.request(`activity=generateSpeech provider=gemini model=${model}`, {
|
|
143
|
+
provider: 'gemini',
|
|
144
|
+
model,
|
|
145
|
+
})
|
|
146
|
+
|
|
147
|
+
const speechConfig: SpeechConfig = {}
|
|
102
148
|
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
149
|
+
if (modelOptions?.multiSpeakerVoiceConfig) {
|
|
150
|
+
// Validate multi-speaker config: 1 or 2 speakers allowed.
|
|
151
|
+
const speakerConfigs =
|
|
152
|
+
modelOptions.multiSpeakerVoiceConfig.speakerVoiceConfigs
|
|
153
|
+
if (
|
|
154
|
+
!Array.isArray(speakerConfigs) ||
|
|
155
|
+
speakerConfigs.length < 1 ||
|
|
156
|
+
speakerConfigs.length > 2
|
|
157
|
+
) {
|
|
158
|
+
throw new Error(
|
|
159
|
+
`Gemini TTS multiSpeakerVoiceConfig.speakerVoiceConfigs must contain 1 or 2 speakers; received ${Array.isArray(speakerConfigs) ? speakerConfigs.length : 'non-array'}.`,
|
|
160
|
+
)
|
|
161
|
+
}
|
|
162
|
+
speechConfig.multiSpeakerVoiceConfig =
|
|
163
|
+
modelOptions.multiSpeakerVoiceConfig
|
|
164
|
+
} else {
|
|
165
|
+
// Honor the standard TTSOptions.voice (used by every other TTS adapter)
|
|
166
|
+
// as a fallback for the prebuilt voice name. If an explicit
|
|
167
|
+
// modelOptions.voiceConfig is supplied its values win — but we still
|
|
168
|
+
// fall back to `voice` / 'Kore' if the supplied voiceConfig is missing
|
|
169
|
+
// prebuiltVoiceConfig.voiceName.
|
|
170
|
+
if (
|
|
171
|
+
voice !== undefined &&
|
|
172
|
+
!(GEMINI_TTS_VOICES as ReadonlyArray<string>).includes(voice)
|
|
173
|
+
) {
|
|
174
|
+
throw new Error(
|
|
175
|
+
`Invalid Gemini TTS voice "${voice}". Valid voices are: ${GEMINI_TTS_VOICES.join(', ')}.`,
|
|
176
|
+
)
|
|
177
|
+
}
|
|
178
|
+
const defaultVoiceName = (voice as GeminiTTSVoice | undefined) ?? 'Kore'
|
|
179
|
+
const supplied = modelOptions?.voiceConfig
|
|
180
|
+
const resolvedVoiceName =
|
|
181
|
+
supplied?.prebuiltVoiceConfig?.voiceName ?? defaultVoiceName
|
|
182
|
+
speechConfig.voiceConfig = {
|
|
183
|
+
prebuiltVoiceConfig: { voiceName: resolvedVoiceName },
|
|
184
|
+
}
|
|
107
185
|
}
|
|
108
186
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
187
|
+
if (modelOptions?.languageCode) {
|
|
188
|
+
speechConfig.languageCode = modelOptions.languageCode
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
try {
|
|
192
|
+
const response = await this.client.models.generateContent({
|
|
193
|
+
model,
|
|
194
|
+
contents: [
|
|
195
|
+
{
|
|
196
|
+
role: 'user',
|
|
197
|
+
parts: [{ text }],
|
|
198
|
+
},
|
|
199
|
+
],
|
|
200
|
+
config: {
|
|
201
|
+
responseModalities: ['AUDIO'],
|
|
202
|
+
speechConfig,
|
|
203
|
+
// systemInstruction belongs inside `config` per the @google/genai
|
|
204
|
+
// contract — matches sibling Gemini adapters (summarize, text).
|
|
205
|
+
...(modelOptions?.systemInstruction && {
|
|
206
|
+
systemInstruction: modelOptions.systemInstruction,
|
|
123
207
|
}),
|
|
124
208
|
},
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
209
|
+
})
|
|
210
|
+
|
|
211
|
+
// Extract audio data from response
|
|
212
|
+
const candidate = response.candidates?.[0]
|
|
213
|
+
const parts = candidate?.content?.parts
|
|
130
214
|
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
215
|
+
if (!parts || parts.length === 0) {
|
|
216
|
+
throw new Error('No audio output received from Gemini TTS')
|
|
217
|
+
}
|
|
134
218
|
|
|
135
|
-
|
|
136
|
-
|
|
219
|
+
// Look for inline data (audio)
|
|
220
|
+
const audioPart = parts.find((part: any) =>
|
|
221
|
+
part.inlineData?.mimeType?.startsWith('audio/'),
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
if (!audioPart || !audioPart.inlineData || !audioPart.inlineData.data) {
|
|
225
|
+
throw new Error('No audio data in Gemini TTS response')
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
const audioBase64 = audioPart.inlineData.data
|
|
229
|
+
// mime is guaranteed by the `startsWith('audio/')` find predicate above.
|
|
230
|
+
const mimeType = audioPart.inlineData.mimeType as string
|
|
231
|
+
|
|
232
|
+
// Gemini TTS models return raw 16-bit LE PCM with a mime type like
|
|
233
|
+
// `audio/L16;codec=pcm;rate=24000`. That isn't playable in an <audio>
|
|
234
|
+
// element and the bare string isn't a usable file extension, so we
|
|
235
|
+
// prepend a RIFF/WAV header here and normalize the result to audio/wav.
|
|
236
|
+
const pcm = parsePcmMimeType(mimeType)
|
|
237
|
+
if (pcm) {
|
|
238
|
+
const wavBase64 = wrapPcmBase64AsWav(
|
|
239
|
+
audioBase64,
|
|
240
|
+
pcm.sampleRate,
|
|
241
|
+
pcm.channels,
|
|
242
|
+
pcm.bitsPerSample,
|
|
243
|
+
)
|
|
244
|
+
return {
|
|
245
|
+
id: generateId(this.name),
|
|
246
|
+
model,
|
|
247
|
+
audio: wavBase64,
|
|
248
|
+
format: 'wav',
|
|
249
|
+
contentType: 'audio/wav',
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
// Strip any mime parameters (e.g. `audio/ogg;codec=opus`) before pulling
|
|
254
|
+
// the subtype out as the file format.
|
|
255
|
+
const format = mimeType.split(';')[0]!.split('/')[1] || 'wav'
|
|
256
|
+
|
|
257
|
+
return {
|
|
258
|
+
id: generateId(this.name),
|
|
259
|
+
model,
|
|
260
|
+
audio: audioBase64,
|
|
261
|
+
format,
|
|
262
|
+
contentType: mimeType,
|
|
263
|
+
}
|
|
264
|
+
} catch (error) {
|
|
265
|
+
logger.errors('gemini.generateSpeech fatal', {
|
|
266
|
+
error,
|
|
267
|
+
source: 'gemini.generateSpeech',
|
|
268
|
+
})
|
|
269
|
+
throw error
|
|
137
270
|
}
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
function parsePcmMimeType(
|
|
275
|
+
mimeType: string,
|
|
276
|
+
): { sampleRate: number; channels: number; bitsPerSample: number } | undefined {
|
|
277
|
+
const normalized = mimeType.toLowerCase()
|
|
278
|
+
const subtype = normalized.split(';')[0]!.split('/')[1] ?? ''
|
|
279
|
+
// Exclude containerized wav (e.g. `audio/wav;codec=pcm`) — those already
|
|
280
|
+
// carry a RIFF header and must not be re-wrapped.
|
|
281
|
+
if (subtype.includes('wav')) return undefined
|
|
282
|
+
|
|
283
|
+
// Accept the variants Gemini and other providers actually emit:
|
|
284
|
+
// - audio/L16;codec=pcm;rate=24000 (IANA PCM with bit depth in the type)
|
|
285
|
+
// - audio/L24 and friends
|
|
286
|
+
// - audio/pcm and audio/x-pcm
|
|
287
|
+
// - anything else that explicitly tags codec=pcm and isn't wav-containered
|
|
288
|
+
const bitDepthMatch = /^audio\/l(\d+)/.exec(normalized)
|
|
289
|
+
const isPcm =
|
|
290
|
+
bitDepthMatch !== null ||
|
|
291
|
+
normalized.startsWith('audio/pcm') ||
|
|
292
|
+
normalized.startsWith('audio/x-pcm') ||
|
|
293
|
+
normalized.includes('codec=pcm')
|
|
294
|
+
if (!isPcm) return undefined
|
|
295
|
+
|
|
296
|
+
const rateMatch = /rate=(\d+)/.exec(normalized)
|
|
297
|
+
const channelsMatch = /channels=(\d+)/.exec(normalized)
|
|
298
|
+
// Default to 16-bit when the mime type doesn't specify — matches Gemini's
|
|
299
|
+
// audio/L16;codec=pcm;rate=24000 response.
|
|
300
|
+
const bitsPerSample = bitDepthMatch ? Number(bitDepthMatch[1]) : 16
|
|
301
|
+
return {
|
|
302
|
+
sampleRate: rateMatch ? Number(rateMatch[1]) : 24000,
|
|
303
|
+
channels: channelsMatch ? Number(channelsMatch[1]) : 1,
|
|
304
|
+
bitsPerSample,
|
|
305
|
+
}
|
|
306
|
+
}
|
|
138
307
|
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
308
|
+
function wrapPcmBase64AsWav(
|
|
309
|
+
pcmBase64: string,
|
|
310
|
+
sampleRate: number,
|
|
311
|
+
channels = 1,
|
|
312
|
+
bitsPerSample = 16,
|
|
313
|
+
): string {
|
|
314
|
+
// The WAV writer below emits a 16-bit PCM fmt chunk. If the source claims a
|
|
315
|
+
// different bit depth we'd be lying about the payload, so bail out loudly
|
|
316
|
+
// rather than producing a corrupt file.
|
|
317
|
+
if (bitsPerSample !== 16) {
|
|
318
|
+
throw new Error(
|
|
319
|
+
`Unsupported PCM bit depth ${bitsPerSample}: only 16-bit PCM can be wrapped as WAV.`,
|
|
142
320
|
)
|
|
321
|
+
}
|
|
143
322
|
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
323
|
+
const pcmBytes =
|
|
324
|
+
typeof Buffer !== 'undefined'
|
|
325
|
+
? new Uint8Array(Buffer.from(pcmBase64, 'base64'))
|
|
326
|
+
: decodeBase64(pcmBase64)
|
|
147
327
|
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
328
|
+
const byteRate = (sampleRate * channels * bitsPerSample) / 8
|
|
329
|
+
const blockAlign = (channels * bitsPerSample) / 8
|
|
330
|
+
const dataSize = pcmBytes.byteLength
|
|
331
|
+
const buffer = new ArrayBuffer(44 + dataSize)
|
|
332
|
+
const view = new DataView(buffer)
|
|
151
333
|
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
334
|
+
writeAscii(view, 0, 'RIFF')
|
|
335
|
+
view.setUint32(4, 36 + dataSize, true)
|
|
336
|
+
writeAscii(view, 8, 'WAVE')
|
|
337
|
+
writeAscii(view, 12, 'fmt ')
|
|
338
|
+
view.setUint32(16, 16, true)
|
|
339
|
+
view.setUint16(20, 1, true)
|
|
340
|
+
view.setUint16(22, channels, true)
|
|
341
|
+
view.setUint32(24, sampleRate, true)
|
|
342
|
+
view.setUint32(28, byteRate, true)
|
|
343
|
+
view.setUint16(32, blockAlign, true)
|
|
344
|
+
view.setUint16(34, bitsPerSample, true)
|
|
345
|
+
writeAscii(view, 36, 'data')
|
|
346
|
+
view.setUint32(40, dataSize, true)
|
|
347
|
+
new Uint8Array(buffer, 44).set(pcmBytes)
|
|
348
|
+
|
|
349
|
+
if (typeof Buffer !== 'undefined') {
|
|
350
|
+
return Buffer.from(buffer).toString('base64')
|
|
351
|
+
}
|
|
352
|
+
let binary = ''
|
|
353
|
+
const bytes = new Uint8Array(buffer)
|
|
354
|
+
for (let i = 0; i < bytes.byteLength; i += 1) {
|
|
355
|
+
binary += String.fromCharCode(bytes[i]!)
|
|
356
|
+
}
|
|
357
|
+
return btoa(binary)
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
function decodeBase64(b64: string): Uint8Array {
|
|
361
|
+
const binary = atob(b64)
|
|
362
|
+
const out = new Uint8Array(binary.length)
|
|
363
|
+
for (let i = 0; i < binary.length; i += 1) out[i] = binary.charCodeAt(i)
|
|
364
|
+
return out
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
function writeAscii(view: DataView, offset: number, text: string): void {
|
|
368
|
+
for (let i = 0; i < text.length; i += 1) {
|
|
369
|
+
view.setUint8(offset + i, text.charCodeAt(i))
|
|
159
370
|
}
|
|
160
371
|
}
|
|
161
372
|
|
|
@@ -185,7 +396,9 @@ export function createGeminiSpeech<TModel extends GeminiTTSModel>(
|
|
|
185
396
|
apiKey: string,
|
|
186
397
|
config?: Omit<GeminiTTSConfig, 'apiKey'>,
|
|
187
398
|
): GeminiTTSAdapter<TModel> {
|
|
188
|
-
|
|
399
|
+
// Put apiKey LAST so caller-supplied config can't silently override the
|
|
400
|
+
// explicit argument.
|
|
401
|
+
return new GeminiTTSAdapter({ ...config, apiKey }, model)
|
|
189
402
|
}
|
|
190
403
|
|
|
191
404
|
/**
|
|
@@ -244,8 +244,28 @@ export function validateImageSize(
|
|
|
244
244
|
}
|
|
245
245
|
|
|
246
246
|
/**
|
|
247
|
-
*
|
|
248
|
-
* Imagen
|
|
247
|
+
* Per-model caps on images per request.
|
|
248
|
+
* Imagen 3 and the Imagen 4 family all support up to 4 images per request
|
|
249
|
+
* via the Gemini API (the rumored 8-image tier is Vertex-only and isn't
|
|
250
|
+
* reachable through @google/genai today). Unknown models fall through to
|
|
251
|
+
* the shared cap defined below.
|
|
252
|
+
*
|
|
253
|
+
* @see https://ai.google.dev/gemini-api/docs/imagen
|
|
254
|
+
*/
|
|
255
|
+
const IMAGEN_MAX_IMAGES_BY_MODEL: Record<string, number> = {
|
|
256
|
+
'imagen-3.0-generate-002': 4,
|
|
257
|
+
'imagen-4.0-generate-001': 4,
|
|
258
|
+
'imagen-4.0-ultra-generate-001': 4,
|
|
259
|
+
'imagen-4.0-fast-generate-001': 4,
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
const DEFAULT_IMAGEN_MAX_IMAGES = 4
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* Validates the number of images requested against the model's known cap.
|
|
266
|
+
* Uses a per-model table where available and falls back to the shared
|
|
267
|
+
* default otherwise — no more "some support up to 8" comments that don't
|
|
268
|
+
* match the error message.
|
|
249
269
|
*/
|
|
250
270
|
export function validateNumberOfImages(
|
|
251
271
|
model: string,
|
|
@@ -253,8 +273,8 @@ export function validateNumberOfImages(
|
|
|
253
273
|
): void {
|
|
254
274
|
if (numberOfImages === undefined) return
|
|
255
275
|
|
|
256
|
-
|
|
257
|
-
|
|
276
|
+
const maxImages =
|
|
277
|
+
IMAGEN_MAX_IMAGES_BY_MODEL[model] ?? DEFAULT_IMAGEN_MAX_IMAGES
|
|
258
278
|
if (numberOfImages < 1 || numberOfImages > maxImages) {
|
|
259
279
|
throw new Error(
|
|
260
280
|
`Invalid numberOfImages "${numberOfImages}" for model "${model}". ` +
|
package/src/index.ts
CHANGED
|
@@ -51,12 +51,26 @@ export {
|
|
|
51
51
|
type GeminiTTSProviderOptions,
|
|
52
52
|
} from './adapters/tts'
|
|
53
53
|
|
|
54
|
+
// Audio / Lyria music generation adapter (experimental)
|
|
55
|
+
/**
|
|
56
|
+
* @experimental Gemini Lyria music generation is an experimental feature and may change.
|
|
57
|
+
*/
|
|
58
|
+
export {
|
|
59
|
+
GeminiAudioAdapter,
|
|
60
|
+
createGeminiAudio,
|
|
61
|
+
geminiAudio,
|
|
62
|
+
type GeminiAudioConfig,
|
|
63
|
+
type GeminiAudioModel,
|
|
64
|
+
type GeminiAudioProviderOptions,
|
|
65
|
+
} from './adapters/audio'
|
|
66
|
+
|
|
54
67
|
// Re-export models from model-meta for convenience
|
|
55
68
|
export { GEMINI_MODELS } from './model-meta'
|
|
56
69
|
export { GEMINI_MODELS as GeminiTextModels } from './model-meta'
|
|
57
70
|
export { GEMINI_IMAGE_MODELS as GeminiImageModels } from './model-meta'
|
|
58
71
|
export { GEMINI_TTS_MODELS as GeminiTTSModels } from './model-meta'
|
|
59
72
|
export { GEMINI_TTS_VOICES as GeminiTTSVoices } from './model-meta'
|
|
73
|
+
export { GEMINI_AUDIO_MODELS as GeminiAudioModels } from './model-meta'
|
|
60
74
|
export type { GeminiModels as GeminiTextModel } from './model-meta'
|
|
61
75
|
export type { GeminiImageModels as GeminiImageModel } from './model-meta'
|
|
62
76
|
export type { GeminiTTSVoice } from './model-meta'
|
package/src/model-meta.ts
CHANGED
|
@@ -295,7 +295,6 @@ const GEMINI_2_5_PRO_TTS = {
|
|
|
295
295
|
input: ['text'],
|
|
296
296
|
output: ['audio'],
|
|
297
297
|
capabilities: ['audio_generation'],
|
|
298
|
-
tools: ['file_search'],
|
|
299
298
|
},
|
|
300
299
|
pricing: {
|
|
301
300
|
input: {
|
|
@@ -455,7 +454,6 @@ const GEMINI_2_5_FLASH_TTS = {
|
|
|
455
454
|
input: ['text'],
|
|
456
455
|
output: ['audio'],
|
|
457
456
|
capabilities: ['audio_generation', 'batch_api'],
|
|
458
|
-
tools: ['file_search'],
|
|
459
457
|
},
|
|
460
458
|
pricing: {
|
|
461
459
|
input: {
|
|
@@ -472,6 +470,82 @@ const GEMINI_2_5_FLASH_TTS = {
|
|
|
472
470
|
GeminiCachedContentOptions
|
|
473
471
|
>
|
|
474
472
|
|
|
473
|
+
/**
|
|
474
|
+
* Gemini 3.1 Flash TTS Preview - latest expressive TTS model with
|
|
475
|
+
* 200+ audio tags, 70+ languages, and multi-speaker dialogue support.
|
|
476
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview
|
|
477
|
+
*/
|
|
478
|
+
const GEMINI_3_1_FLASH_TTS = {
|
|
479
|
+
name: 'gemini-3.1-flash-tts-preview',
|
|
480
|
+
max_input_tokens: 32_768,
|
|
481
|
+
max_output_tokens: 16_384,
|
|
482
|
+
knowledge_cutoff: '2025-05-01',
|
|
483
|
+
supports: {
|
|
484
|
+
input: ['text'],
|
|
485
|
+
output: ['audio'],
|
|
486
|
+
capabilities: ['audio_generation', 'batch_api'],
|
|
487
|
+
},
|
|
488
|
+
pricing: {
|
|
489
|
+
input: {
|
|
490
|
+
normal: 0.5,
|
|
491
|
+
},
|
|
492
|
+
output: {
|
|
493
|
+
normal: 10,
|
|
494
|
+
},
|
|
495
|
+
},
|
|
496
|
+
} as const satisfies ModelMeta<
|
|
497
|
+
GeminiToolConfigOptions &
|
|
498
|
+
GeminiSafetyOptions &
|
|
499
|
+
GeminiCommonConfigOptions &
|
|
500
|
+
GeminiCachedContentOptions
|
|
501
|
+
>
|
|
502
|
+
|
|
503
|
+
/**
|
|
504
|
+
* Lyria 3 Pro Preview — Google's flagship music generation model.
|
|
505
|
+
* Generates full-length songs with multiple verses, choruses, and bridges.
|
|
506
|
+
* Outputs MP3 or WAV at 48 kHz stereo.
|
|
507
|
+
* @see https://ai.google.dev/gemini-api/docs/models/lyria-3-pro-preview
|
|
508
|
+
*/
|
|
509
|
+
const LYRIA_3_PRO = {
|
|
510
|
+
name: 'lyria-3-pro-preview',
|
|
511
|
+
max_input_tokens: 131_072,
|
|
512
|
+
supports: {
|
|
513
|
+
input: ['text', 'image'],
|
|
514
|
+
output: ['audio'],
|
|
515
|
+
capabilities: ['audio_generation'],
|
|
516
|
+
},
|
|
517
|
+
pricing: {
|
|
518
|
+
input: {
|
|
519
|
+
normal: 0,
|
|
520
|
+
},
|
|
521
|
+
output: {
|
|
522
|
+
normal: 0,
|
|
523
|
+
},
|
|
524
|
+
},
|
|
525
|
+
} as const satisfies ModelMeta
|
|
526
|
+
|
|
527
|
+
/**
|
|
528
|
+
* Lyria 3 Clip Preview — 30-second music clips in MP3 format.
|
|
529
|
+
* @see https://ai.google.dev/gemini-api/docs/music-generation
|
|
530
|
+
*/
|
|
531
|
+
const LYRIA_3_CLIP = {
|
|
532
|
+
name: 'lyria-3-clip-preview',
|
|
533
|
+
max_input_tokens: 131_072,
|
|
534
|
+
supports: {
|
|
535
|
+
input: ['text', 'image'],
|
|
536
|
+
output: ['audio'],
|
|
537
|
+
capabilities: ['audio_generation'],
|
|
538
|
+
},
|
|
539
|
+
pricing: {
|
|
540
|
+
input: {
|
|
541
|
+
normal: 0,
|
|
542
|
+
},
|
|
543
|
+
output: {
|
|
544
|
+
normal: 0,
|
|
545
|
+
},
|
|
546
|
+
},
|
|
547
|
+
} as const satisfies ModelMeta
|
|
548
|
+
|
|
475
549
|
const GEMINI_2_5_FLASH_LITE = {
|
|
476
550
|
name: 'gemini-2.5-flash-lite',
|
|
477
551
|
max_input_tokens: 1_048_576,
|
|
@@ -580,7 +654,7 @@ const GEMINI_2_FLASH_IMAGE = {
|
|
|
580
654
|
knowledge_cutoff: '2024-08-01',
|
|
581
655
|
supports: {
|
|
582
656
|
input: ['text', 'image', 'audio', 'video'],
|
|
583
|
-
output: ['text'],
|
|
657
|
+
output: ['text', 'image'],
|
|
584
658
|
capabilities: ['batch_api', 'caching', 'structured_output'],
|
|
585
659
|
tools: [],
|
|
586
660
|
},
|
|
@@ -930,10 +1004,20 @@ export const GEMINI_IMAGE_MODELS = [
|
|
|
930
1004
|
* @experimental Gemini TTS is an experimental feature and may change.
|
|
931
1005
|
*/
|
|
932
1006
|
export const GEMINI_TTS_MODELS = [
|
|
1007
|
+
GEMINI_3_1_FLASH_TTS.name,
|
|
933
1008
|
GEMINI_2_5_FLASH_TTS.name,
|
|
934
1009
|
GEMINI_2_5_PRO_TTS.name,
|
|
935
1010
|
] as const
|
|
936
1011
|
|
|
1012
|
+
/**
|
|
1013
|
+
* Audio generation models (Lyria music generation).
|
|
1014
|
+
* @experimental Lyria music generation is an experimental feature and may change.
|
|
1015
|
+
*/
|
|
1016
|
+
export const GEMINI_AUDIO_MODELS = [
|
|
1017
|
+
LYRIA_3_PRO.name,
|
|
1018
|
+
LYRIA_3_CLIP.name,
|
|
1019
|
+
] as const
|
|
1020
|
+
|
|
937
1021
|
/**
|
|
938
1022
|
* Available voice names for Gemini TTS
|
|
939
1023
|
* @see https://ai.google.dev/gemini-api/docs/speech-generation
|