@tanstack/ai-gemini 0.9.1 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/audio.d.ts +92 -0
- package/dist/esm/adapters/audio.js +61 -0
- package/dist/esm/adapters/audio.js.map +1 -0
- package/dist/esm/adapters/image.js +42 -10
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/tts.d.ts +39 -3
- package/dist/esm/adapters/tts.js +114 -18
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/image/image-provider-options.d.ts +4 -2
- package/dist/esm/image/image-provider-options.js +8 -1
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +5 -0
- package/dist/esm/index.js +6 -1
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/model-meta.d.ts +6 -1
- package/dist/esm/model-meta.js +15 -0
- package/dist/esm/model-meta.js.map +1 -1
- package/package.json +3 -3
- package/src/adapters/audio.ts +185 -0
- package/src/adapters/image.ts +70 -9
- package/src/adapters/tts.ts +222 -23
- package/src/image/image-provider-options.ts +24 -4
- package/src/index.ts +14 -0
- package/src/model-meta.ts +87 -3
package/src/adapters/image.ts
CHANGED
|
@@ -74,7 +74,7 @@ export class GeminiImageAdapter<
|
|
|
74
74
|
private client: GoogleGenAI
|
|
75
75
|
|
|
76
76
|
constructor(config: GeminiImageConfig, model: TModel) {
|
|
77
|
-
super(
|
|
77
|
+
super(model, config)
|
|
78
78
|
this.client = createGeminiClient(config)
|
|
79
79
|
}
|
|
80
80
|
|
|
@@ -138,8 +138,24 @@ export class GeminiImageAdapter<
|
|
|
138
138
|
? `${prompt} Generate ${numberOfImages} distinct images.`
|
|
139
139
|
: prompt
|
|
140
140
|
|
|
141
|
+
// GeminiImageProviderOptions is Imagen-shaped — most fields
|
|
142
|
+
// (personGeneration, safetyFilterLevel, addWatermark, outputMimeType,
|
|
143
|
+
// outputCompressionQuality, guidanceScale, enhancePrompt,
|
|
144
|
+
// includeSafetyAttributes, includeRaiReason, outputGcsUri, labels,
|
|
145
|
+
// negativePrompt, language) are only valid on GenerateImagesConfig and
|
|
146
|
+
// would be rejected by the Gemini-native generateContent path. Pick only
|
|
147
|
+
// the fields that are valid on GenerateContentConfig instead of spreading
|
|
148
|
+
// the whole options object.
|
|
149
|
+
const nativeConfig: GenerateContentConfig = {}
|
|
150
|
+
if (modelOptions?.seed !== undefined) {
|
|
151
|
+
nativeConfig.seed = modelOptions.seed
|
|
152
|
+
}
|
|
153
|
+
|
|
141
154
|
const config: GenerateContentConfig = {
|
|
142
|
-
|
|
155
|
+
...nativeConfig,
|
|
156
|
+
// Include TEXT so the model can interleave descriptions between images.
|
|
157
|
+
// IMPORTANT: responseModalities is a protected default — set it AFTER
|
|
158
|
+
// nativeConfig so nothing can silently disable image output.
|
|
143
159
|
responseModalities: ['TEXT', 'IMAGE'],
|
|
144
160
|
...(parsedSize && {
|
|
145
161
|
imageConfig: {
|
|
@@ -151,7 +167,6 @@ export class GeminiImageAdapter<
|
|
|
151
167
|
}),
|
|
152
168
|
},
|
|
153
169
|
}),
|
|
154
|
-
...modelOptions,
|
|
155
170
|
}
|
|
156
171
|
|
|
157
172
|
const response = await this.client.models.generateContent({
|
|
@@ -168,6 +183,7 @@ export class GeminiImageAdapter<
|
|
|
168
183
|
response: GenerateContentResponse,
|
|
169
184
|
): ImageGenerationResult {
|
|
170
185
|
const images: Array<GeneratedImage> = []
|
|
186
|
+
const textParts: Array<string> = []
|
|
171
187
|
const parts = response.candidates?.[0]?.content?.parts ?? []
|
|
172
188
|
|
|
173
189
|
for (const part of parts) {
|
|
@@ -177,9 +193,23 @@ export class GeminiImageAdapter<
|
|
|
177
193
|
part.inlineData.data.length > 0
|
|
178
194
|
) {
|
|
179
195
|
images.push({ b64Json: part.inlineData.data })
|
|
196
|
+
} else if (typeof part.text === 'string' && part.text.length > 0) {
|
|
197
|
+
textParts.push(part.text)
|
|
180
198
|
}
|
|
181
199
|
}
|
|
182
200
|
|
|
201
|
+
// If the model returned only text parts (for example a safety refusal
|
|
202
|
+
// or a "can't do that" message), surface the text instead of silently
|
|
203
|
+
// resolving to an empty images array — otherwise callers can't tell a
|
|
204
|
+
// generation failure apart from a genuine empty response.
|
|
205
|
+
if (images.length === 0) {
|
|
206
|
+
const reason =
|
|
207
|
+
textParts.length > 0
|
|
208
|
+
? `: ${textParts.join(' ').trim()}`
|
|
209
|
+
: ' (no inline image or text parts were returned).'
|
|
210
|
+
throw new Error(`Gemini ${model} returned no images${reason}`)
|
|
211
|
+
}
|
|
212
|
+
|
|
183
213
|
return {
|
|
184
214
|
id: generateId(this.name),
|
|
185
215
|
model,
|
|
@@ -205,12 +235,43 @@ export class GeminiImageAdapter<
|
|
|
205
235
|
model: string,
|
|
206
236
|
response: GenerateImagesResponse,
|
|
207
237
|
): ImageGenerationResult {
|
|
208
|
-
const
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
238
|
+
const entries = response.generatedImages ?? []
|
|
239
|
+
const images: Array<GeneratedImage> = []
|
|
240
|
+
const filterReasons: Array<string> = []
|
|
241
|
+
|
|
242
|
+
for (const item of entries) {
|
|
243
|
+
const b64Json = item.image?.imageBytes
|
|
244
|
+
if (b64Json) {
|
|
245
|
+
images.push({ b64Json, revisedPrompt: item.enhancedPrompt })
|
|
246
|
+
continue
|
|
247
|
+
}
|
|
248
|
+
// Imagen can drop individual entries with a raiFilteredReason when
|
|
249
|
+
// Responsible-AI filters fire. Preserve the reason so callers can
|
|
250
|
+
// surface it instead of silently getting back fewer images.
|
|
251
|
+
const reason = (item as { raiFilteredReason?: string }).raiFilteredReason
|
|
252
|
+
if (reason) {
|
|
253
|
+
filterReasons.push(reason)
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
// Every entry was filtered — no usable images to return. Throw rather
|
|
258
|
+
// than resolve to an empty array so the caller is forced to handle the
|
|
259
|
+
// failure mode explicitly.
|
|
260
|
+
if (entries.length > 0 && images.length === 0) {
|
|
261
|
+
const joined = filterReasons.length > 0 ? filterReasons.join('; ') : ''
|
|
262
|
+
throw new Error(
|
|
263
|
+
`Imagen ${model} returned no images: all ${entries.length} generated image(s) were filtered by Responsible-AI${joined ? ` (${joined})` : ''}.`,
|
|
264
|
+
)
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
// Partial filter: surface via console.warn since ImageGenerationResult
|
|
268
|
+
// has no warnings field. Callers that care can still inspect the count
|
|
269
|
+
// mismatch between requested and returned images.
|
|
270
|
+
if (filterReasons.length > 0 && typeof console !== 'undefined') {
|
|
271
|
+
console.warn(
|
|
272
|
+
`[gemini-image] ${filterReasons.length} of ${entries.length} images from ${model} were filtered by Responsible-AI: ${filterReasons.join('; ')}`,
|
|
273
|
+
)
|
|
274
|
+
}
|
|
214
275
|
|
|
215
276
|
return {
|
|
216
277
|
id: generateId(this.name),
|
package/src/adapters/tts.ts
CHANGED
|
@@ -4,21 +4,40 @@ import {
|
|
|
4
4
|
generateId,
|
|
5
5
|
getGeminiApiKeyFromEnv,
|
|
6
6
|
} from '../utils'
|
|
7
|
+
import { GEMINI_TTS_VOICES } from '../model-meta'
|
|
7
8
|
import type { GEMINI_TTS_MODELS, GeminiTTSVoice } from '../model-meta'
|
|
8
9
|
import type { TTSOptions, TTSResult } from '@tanstack/ai'
|
|
9
|
-
import type { GoogleGenAI } from '@google/genai'
|
|
10
|
+
import type { GoogleGenAI, SpeechConfig } from '@google/genai'
|
|
10
11
|
import type { GeminiClientConfig } from '../utils'
|
|
11
12
|
|
|
13
|
+
/**
|
|
14
|
+
* Configuration for a single speaker in a multi-speaker dialogue.
|
|
15
|
+
* Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.
|
|
16
|
+
*/
|
|
17
|
+
export interface GeminiSpeakerVoiceConfig {
|
|
18
|
+
/** A name used in the prompt to refer to this speaker */
|
|
19
|
+
speaker: string
|
|
20
|
+
/** Voice configuration for this speaker */
|
|
21
|
+
voiceConfig: {
|
|
22
|
+
prebuiltVoiceConfig: {
|
|
23
|
+
voiceName: GeminiTTSVoice
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
|
|
12
28
|
/**
|
|
13
29
|
* Provider-specific options for Gemini TTS
|
|
14
30
|
*
|
|
15
31
|
* @experimental Gemini TTS is an experimental feature.
|
|
16
32
|
* @see https://ai.google.dev/gemini-api/docs/speech-generation
|
|
33
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview
|
|
17
34
|
*/
|
|
18
35
|
export interface GeminiTTSProviderOptions {
|
|
19
36
|
/**
|
|
20
|
-
* Voice configuration for TTS.
|
|
37
|
+
* Voice configuration for single-speaker TTS.
|
|
21
38
|
* Choose from 30 available voices with different characteristics.
|
|
39
|
+
*
|
|
40
|
+
* Use `multiSpeakerVoiceConfig` instead for dialogues.
|
|
22
41
|
*/
|
|
23
42
|
voiceConfig?: {
|
|
24
43
|
prebuiltVoiceConfig?: {
|
|
@@ -30,11 +49,31 @@ export interface GeminiTTSProviderOptions {
|
|
|
30
49
|
}
|
|
31
50
|
}
|
|
32
51
|
|
|
52
|
+
/**
|
|
53
|
+
* Multi-speaker voice configuration (up to 2 speakers).
|
|
54
|
+
* Supported by Gemini 3.1 Flash TTS Preview and the 2.5 TTS models.
|
|
55
|
+
*
|
|
56
|
+
* Each speaker's lines in the prompt are prefixed with the name defined
|
|
57
|
+
* here, e.g.:
|
|
58
|
+
*
|
|
59
|
+
* ```text
|
|
60
|
+
* Joe: Hey, how's it going?
|
|
61
|
+
* Jane: Not bad, you?
|
|
62
|
+
* ```
|
|
63
|
+
*/
|
|
64
|
+
multiSpeakerVoiceConfig?: {
|
|
65
|
+
speakerVoiceConfigs: Array<GeminiSpeakerVoiceConfig>
|
|
66
|
+
}
|
|
67
|
+
|
|
33
68
|
/**
|
|
34
69
|
* System instruction for controlling speech style.
|
|
35
70
|
* Use natural language to describe the desired speaking style,
|
|
36
71
|
* pace, tone, accent, or other characteristics.
|
|
37
72
|
*
|
|
73
|
+
* With Gemini 3.1 Flash TTS, you can also use inline audio tags like
|
|
74
|
+
* `[whispering]`, `[laughs]`, `[excited]` directly in the input text
|
|
75
|
+
* to control delivery.
|
|
76
|
+
*
|
|
38
77
|
* @example "Speak slowly and calmly, as if telling a bedtime story"
|
|
39
78
|
* @example "Use an upbeat, enthusiastic tone with moderate pace"
|
|
40
79
|
* @example "Speak with a British accent"
|
|
@@ -43,8 +82,8 @@ export interface GeminiTTSProviderOptions {
|
|
|
43
82
|
|
|
44
83
|
/**
|
|
45
84
|
* Language code hint for the speech synthesis.
|
|
46
|
-
* Gemini TTS supports
|
|
47
|
-
*
|
|
85
|
+
* Gemini 3.1 Flash TTS supports 70+ languages with auto-detection;
|
|
86
|
+
* the 2.5 TTS models support 24 languages.
|
|
48
87
|
*
|
|
49
88
|
* @example "en-US" for American English
|
|
50
89
|
* @example "es-ES" for Spanish (Spain)
|
|
@@ -85,7 +124,7 @@ export class GeminiTTSAdapter<
|
|
|
85
124
|
private client: GoogleGenAI
|
|
86
125
|
|
|
87
126
|
constructor(config: GeminiTTSConfig, model: TModel) {
|
|
88
|
-
super(
|
|
127
|
+
super(model, config)
|
|
89
128
|
this.client = createGeminiClient(config)
|
|
90
129
|
}
|
|
91
130
|
|
|
@@ -98,18 +137,55 @@ export class GeminiTTSAdapter<
|
|
|
98
137
|
async generateSpeech(
|
|
99
138
|
options: TTSOptions<GeminiTTSProviderOptions>,
|
|
100
139
|
): Promise<TTSResult> {
|
|
101
|
-
const { logger } = options
|
|
102
|
-
const { model, text, modelOptions } = options
|
|
140
|
+
const { model, text, modelOptions, voice, logger } = options
|
|
103
141
|
|
|
104
142
|
logger.request(`activity=generateSpeech provider=gemini model=${model}`, {
|
|
105
143
|
provider: 'gemini',
|
|
106
144
|
model,
|
|
107
145
|
})
|
|
108
146
|
|
|
109
|
-
const
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
147
|
+
const speechConfig: SpeechConfig = {}
|
|
148
|
+
|
|
149
|
+
if (modelOptions?.multiSpeakerVoiceConfig) {
|
|
150
|
+
// Validate multi-speaker config: 1 or 2 speakers allowed.
|
|
151
|
+
const speakerConfigs =
|
|
152
|
+
modelOptions.multiSpeakerVoiceConfig.speakerVoiceConfigs
|
|
153
|
+
if (
|
|
154
|
+
!Array.isArray(speakerConfigs) ||
|
|
155
|
+
speakerConfigs.length < 1 ||
|
|
156
|
+
speakerConfigs.length > 2
|
|
157
|
+
) {
|
|
158
|
+
throw new Error(
|
|
159
|
+
`Gemini TTS multiSpeakerVoiceConfig.speakerVoiceConfigs must contain 1 or 2 speakers; received ${Array.isArray(speakerConfigs) ? speakerConfigs.length : 'non-array'}.`,
|
|
160
|
+
)
|
|
161
|
+
}
|
|
162
|
+
speechConfig.multiSpeakerVoiceConfig =
|
|
163
|
+
modelOptions.multiSpeakerVoiceConfig
|
|
164
|
+
} else {
|
|
165
|
+
// Honor the standard TTSOptions.voice (used by every other TTS adapter)
|
|
166
|
+
// as a fallback for the prebuilt voice name. If an explicit
|
|
167
|
+
// modelOptions.voiceConfig is supplied its values win — but we still
|
|
168
|
+
// fall back to `voice` / 'Kore' if the supplied voiceConfig is missing
|
|
169
|
+
// prebuiltVoiceConfig.voiceName.
|
|
170
|
+
if (
|
|
171
|
+
voice !== undefined &&
|
|
172
|
+
!(GEMINI_TTS_VOICES as ReadonlyArray<string>).includes(voice)
|
|
173
|
+
) {
|
|
174
|
+
throw new Error(
|
|
175
|
+
`Invalid Gemini TTS voice "${voice}". Valid voices are: ${GEMINI_TTS_VOICES.join(', ')}.`,
|
|
176
|
+
)
|
|
177
|
+
}
|
|
178
|
+
const defaultVoiceName = (voice as GeminiTTSVoice | undefined) ?? 'Kore'
|
|
179
|
+
const supplied = modelOptions?.voiceConfig
|
|
180
|
+
const resolvedVoiceName =
|
|
181
|
+
supplied?.prebuiltVoiceConfig?.voiceName ?? defaultVoiceName
|
|
182
|
+
speechConfig.voiceConfig = {
|
|
183
|
+
prebuiltVoiceConfig: { voiceName: resolvedVoiceName },
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
if (modelOptions?.languageCode) {
|
|
188
|
+
speechConfig.languageCode = modelOptions.languageCode
|
|
113
189
|
}
|
|
114
190
|
|
|
115
191
|
try {
|
|
@@ -123,16 +199,13 @@ export class GeminiTTSAdapter<
|
|
|
123
199
|
],
|
|
124
200
|
config: {
|
|
125
201
|
responseModalities: ['AUDIO'],
|
|
126
|
-
speechConfig
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
},
|
|
202
|
+
speechConfig,
|
|
203
|
+
// systemInstruction belongs inside `config` per the @google/genai
|
|
204
|
+
// contract — matches sibling Gemini adapters (summarize, text).
|
|
205
|
+
...(modelOptions?.systemInstruction && {
|
|
206
|
+
systemInstruction: modelOptions.systemInstruction,
|
|
207
|
+
}),
|
|
132
208
|
},
|
|
133
|
-
...(modelOptions?.systemInstruction && {
|
|
134
|
-
systemInstruction: modelOptions.systemInstruction,
|
|
135
|
-
}),
|
|
136
209
|
})
|
|
137
210
|
|
|
138
211
|
// Extract audio data from response
|
|
@@ -153,8 +226,33 @@ export class GeminiTTSAdapter<
|
|
|
153
226
|
}
|
|
154
227
|
|
|
155
228
|
const audioBase64 = audioPart.inlineData.data
|
|
156
|
-
|
|
157
|
-
const
|
|
229
|
+
// mime is guaranteed by the `startsWith('audio/')` find predicate above.
|
|
230
|
+
const mimeType = audioPart.inlineData.mimeType as string
|
|
231
|
+
|
|
232
|
+
// Gemini TTS models return raw 16-bit LE PCM with a mime type like
|
|
233
|
+
// `audio/L16;codec=pcm;rate=24000`. That isn't playable in an <audio>
|
|
234
|
+
// element and the bare string isn't a usable file extension, so we
|
|
235
|
+
// prepend a RIFF/WAV header here and normalize the result to audio/wav.
|
|
236
|
+
const pcm = parsePcmMimeType(mimeType)
|
|
237
|
+
if (pcm) {
|
|
238
|
+
const wavBase64 = wrapPcmBase64AsWav(
|
|
239
|
+
audioBase64,
|
|
240
|
+
pcm.sampleRate,
|
|
241
|
+
pcm.channels,
|
|
242
|
+
pcm.bitsPerSample,
|
|
243
|
+
)
|
|
244
|
+
return {
|
|
245
|
+
id: generateId(this.name),
|
|
246
|
+
model,
|
|
247
|
+
audio: wavBase64,
|
|
248
|
+
format: 'wav',
|
|
249
|
+
contentType: 'audio/wav',
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
// Strip any mime parameters (e.g. `audio/ogg;codec=opus`) before pulling
|
|
254
|
+
// the subtype out as the file format.
|
|
255
|
+
const format = mimeType.split(';')[0]!.split('/')[1] || 'wav'
|
|
158
256
|
|
|
159
257
|
return {
|
|
160
258
|
id: generateId(this.name),
|
|
@@ -173,6 +271,105 @@ export class GeminiTTSAdapter<
|
|
|
173
271
|
}
|
|
174
272
|
}
|
|
175
273
|
|
|
274
|
+
function parsePcmMimeType(
|
|
275
|
+
mimeType: string,
|
|
276
|
+
): { sampleRate: number; channels: number; bitsPerSample: number } | undefined {
|
|
277
|
+
const normalized = mimeType.toLowerCase()
|
|
278
|
+
const subtype = normalized.split(';')[0]!.split('/')[1] ?? ''
|
|
279
|
+
// Exclude containerized wav (e.g. `audio/wav;codec=pcm`) — those already
|
|
280
|
+
// carry a RIFF header and must not be re-wrapped.
|
|
281
|
+
if (subtype.includes('wav')) return undefined
|
|
282
|
+
|
|
283
|
+
// Accept the variants Gemini and other providers actually emit:
|
|
284
|
+
// - audio/L16;codec=pcm;rate=24000 (IANA PCM with bit depth in the type)
|
|
285
|
+
// - audio/L24 and friends
|
|
286
|
+
// - audio/pcm and audio/x-pcm
|
|
287
|
+
// - anything else that explicitly tags codec=pcm and isn't wav-containered
|
|
288
|
+
const bitDepthMatch = /^audio\/l(\d+)/.exec(normalized)
|
|
289
|
+
const isPcm =
|
|
290
|
+
bitDepthMatch !== null ||
|
|
291
|
+
normalized.startsWith('audio/pcm') ||
|
|
292
|
+
normalized.startsWith('audio/x-pcm') ||
|
|
293
|
+
normalized.includes('codec=pcm')
|
|
294
|
+
if (!isPcm) return undefined
|
|
295
|
+
|
|
296
|
+
const rateMatch = /rate=(\d+)/.exec(normalized)
|
|
297
|
+
const channelsMatch = /channels=(\d+)/.exec(normalized)
|
|
298
|
+
// Default to 16-bit when the mime type doesn't specify — matches Gemini's
|
|
299
|
+
// audio/L16;codec=pcm;rate=24000 response.
|
|
300
|
+
const bitsPerSample = bitDepthMatch ? Number(bitDepthMatch[1]) : 16
|
|
301
|
+
return {
|
|
302
|
+
sampleRate: rateMatch ? Number(rateMatch[1]) : 24000,
|
|
303
|
+
channels: channelsMatch ? Number(channelsMatch[1]) : 1,
|
|
304
|
+
bitsPerSample,
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
function wrapPcmBase64AsWav(
|
|
309
|
+
pcmBase64: string,
|
|
310
|
+
sampleRate: number,
|
|
311
|
+
channels = 1,
|
|
312
|
+
bitsPerSample = 16,
|
|
313
|
+
): string {
|
|
314
|
+
// The WAV writer below emits a 16-bit PCM fmt chunk. If the source claims a
|
|
315
|
+
// different bit depth we'd be lying about the payload, so bail out loudly
|
|
316
|
+
// rather than producing a corrupt file.
|
|
317
|
+
if (bitsPerSample !== 16) {
|
|
318
|
+
throw new Error(
|
|
319
|
+
`Unsupported PCM bit depth ${bitsPerSample}: only 16-bit PCM can be wrapped as WAV.`,
|
|
320
|
+
)
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
const pcmBytes =
|
|
324
|
+
typeof Buffer !== 'undefined'
|
|
325
|
+
? new Uint8Array(Buffer.from(pcmBase64, 'base64'))
|
|
326
|
+
: decodeBase64(pcmBase64)
|
|
327
|
+
|
|
328
|
+
const byteRate = (sampleRate * channels * bitsPerSample) / 8
|
|
329
|
+
const blockAlign = (channels * bitsPerSample) / 8
|
|
330
|
+
const dataSize = pcmBytes.byteLength
|
|
331
|
+
const buffer = new ArrayBuffer(44 + dataSize)
|
|
332
|
+
const view = new DataView(buffer)
|
|
333
|
+
|
|
334
|
+
writeAscii(view, 0, 'RIFF')
|
|
335
|
+
view.setUint32(4, 36 + dataSize, true)
|
|
336
|
+
writeAscii(view, 8, 'WAVE')
|
|
337
|
+
writeAscii(view, 12, 'fmt ')
|
|
338
|
+
view.setUint32(16, 16, true)
|
|
339
|
+
view.setUint16(20, 1, true)
|
|
340
|
+
view.setUint16(22, channels, true)
|
|
341
|
+
view.setUint32(24, sampleRate, true)
|
|
342
|
+
view.setUint32(28, byteRate, true)
|
|
343
|
+
view.setUint16(32, blockAlign, true)
|
|
344
|
+
view.setUint16(34, bitsPerSample, true)
|
|
345
|
+
writeAscii(view, 36, 'data')
|
|
346
|
+
view.setUint32(40, dataSize, true)
|
|
347
|
+
new Uint8Array(buffer, 44).set(pcmBytes)
|
|
348
|
+
|
|
349
|
+
if (typeof Buffer !== 'undefined') {
|
|
350
|
+
return Buffer.from(buffer).toString('base64')
|
|
351
|
+
}
|
|
352
|
+
let binary = ''
|
|
353
|
+
const bytes = new Uint8Array(buffer)
|
|
354
|
+
for (let i = 0; i < bytes.byteLength; i += 1) {
|
|
355
|
+
binary += String.fromCharCode(bytes[i]!)
|
|
356
|
+
}
|
|
357
|
+
return btoa(binary)
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
function decodeBase64(b64: string): Uint8Array {
|
|
361
|
+
const binary = atob(b64)
|
|
362
|
+
const out = new Uint8Array(binary.length)
|
|
363
|
+
for (let i = 0; i < binary.length; i += 1) out[i] = binary.charCodeAt(i)
|
|
364
|
+
return out
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
function writeAscii(view: DataView, offset: number, text: string): void {
|
|
368
|
+
for (let i = 0; i < text.length; i += 1) {
|
|
369
|
+
view.setUint8(offset + i, text.charCodeAt(i))
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
|
|
176
373
|
/**
|
|
177
374
|
* Creates a Gemini TTS adapter with explicit API key.
|
|
178
375
|
* Type resolution happens here at the call site.
|
|
@@ -199,7 +396,9 @@ export function createGeminiSpeech<TModel extends GeminiTTSModel>(
|
|
|
199
396
|
apiKey: string,
|
|
200
397
|
config?: Omit<GeminiTTSConfig, 'apiKey'>,
|
|
201
398
|
): GeminiTTSAdapter<TModel> {
|
|
202
|
-
|
|
399
|
+
// Put apiKey LAST so caller-supplied config can't silently override the
|
|
400
|
+
// explicit argument.
|
|
401
|
+
return new GeminiTTSAdapter({ ...config, apiKey }, model)
|
|
203
402
|
}
|
|
204
403
|
|
|
205
404
|
/**
|
|
@@ -244,8 +244,28 @@ export function validateImageSize(
|
|
|
244
244
|
}
|
|
245
245
|
|
|
246
246
|
/**
|
|
247
|
-
*
|
|
248
|
-
* Imagen
|
|
247
|
+
* Per-model caps on images per request.
|
|
248
|
+
* Imagen 3 and the Imagen 4 family all support up to 4 images per request
|
|
249
|
+
* via the Gemini API (the rumored 8-image tier is Vertex-only and isn't
|
|
250
|
+
* reachable through @google/genai today). Unknown models fall through to
|
|
251
|
+
* the shared cap defined below.
|
|
252
|
+
*
|
|
253
|
+
* @see https://ai.google.dev/gemini-api/docs/imagen
|
|
254
|
+
*/
|
|
255
|
+
const IMAGEN_MAX_IMAGES_BY_MODEL: Record<string, number> = {
|
|
256
|
+
'imagen-3.0-generate-002': 4,
|
|
257
|
+
'imagen-4.0-generate-001': 4,
|
|
258
|
+
'imagen-4.0-ultra-generate-001': 4,
|
|
259
|
+
'imagen-4.0-fast-generate-001': 4,
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
const DEFAULT_IMAGEN_MAX_IMAGES = 4
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* Validates the number of images requested against the model's known cap.
|
|
266
|
+
* Uses a per-model table where available and falls back to the shared
|
|
267
|
+
* default otherwise — no more "some support up to 8" comments that don't
|
|
268
|
+
* match the error message.
|
|
249
269
|
*/
|
|
250
270
|
export function validateNumberOfImages(
|
|
251
271
|
model: string,
|
|
@@ -253,8 +273,8 @@ export function validateNumberOfImages(
|
|
|
253
273
|
): void {
|
|
254
274
|
if (numberOfImages === undefined) return
|
|
255
275
|
|
|
256
|
-
|
|
257
|
-
|
|
276
|
+
const maxImages =
|
|
277
|
+
IMAGEN_MAX_IMAGES_BY_MODEL[model] ?? DEFAULT_IMAGEN_MAX_IMAGES
|
|
258
278
|
if (numberOfImages < 1 || numberOfImages > maxImages) {
|
|
259
279
|
throw new Error(
|
|
260
280
|
`Invalid numberOfImages "${numberOfImages}" for model "${model}". ` +
|
package/src/index.ts
CHANGED
|
@@ -51,12 +51,26 @@ export {
|
|
|
51
51
|
type GeminiTTSProviderOptions,
|
|
52
52
|
} from './adapters/tts'
|
|
53
53
|
|
|
54
|
+
// Audio / Lyria music generation adapter (experimental)
|
|
55
|
+
/**
|
|
56
|
+
* @experimental Gemini Lyria music generation is an experimental feature and may change.
|
|
57
|
+
*/
|
|
58
|
+
export {
|
|
59
|
+
GeminiAudioAdapter,
|
|
60
|
+
createGeminiAudio,
|
|
61
|
+
geminiAudio,
|
|
62
|
+
type GeminiAudioConfig,
|
|
63
|
+
type GeminiAudioModel,
|
|
64
|
+
type GeminiAudioProviderOptions,
|
|
65
|
+
} from './adapters/audio'
|
|
66
|
+
|
|
54
67
|
// Re-export models from model-meta for convenience
|
|
55
68
|
export { GEMINI_MODELS } from './model-meta'
|
|
56
69
|
export { GEMINI_MODELS as GeminiTextModels } from './model-meta'
|
|
57
70
|
export { GEMINI_IMAGE_MODELS as GeminiImageModels } from './model-meta'
|
|
58
71
|
export { GEMINI_TTS_MODELS as GeminiTTSModels } from './model-meta'
|
|
59
72
|
export { GEMINI_TTS_VOICES as GeminiTTSVoices } from './model-meta'
|
|
73
|
+
export { GEMINI_AUDIO_MODELS as GeminiAudioModels } from './model-meta'
|
|
60
74
|
export type { GeminiModels as GeminiTextModel } from './model-meta'
|
|
61
75
|
export type { GeminiImageModels as GeminiImageModel } from './model-meta'
|
|
62
76
|
export type { GeminiTTSVoice } from './model-meta'
|
package/src/model-meta.ts
CHANGED
|
@@ -295,7 +295,6 @@ const GEMINI_2_5_PRO_TTS = {
|
|
|
295
295
|
input: ['text'],
|
|
296
296
|
output: ['audio'],
|
|
297
297
|
capabilities: ['audio_generation'],
|
|
298
|
-
tools: ['file_search'],
|
|
299
298
|
},
|
|
300
299
|
pricing: {
|
|
301
300
|
input: {
|
|
@@ -455,7 +454,6 @@ const GEMINI_2_5_FLASH_TTS = {
|
|
|
455
454
|
input: ['text'],
|
|
456
455
|
output: ['audio'],
|
|
457
456
|
capabilities: ['audio_generation', 'batch_api'],
|
|
458
|
-
tools: ['file_search'],
|
|
459
457
|
},
|
|
460
458
|
pricing: {
|
|
461
459
|
input: {
|
|
@@ -472,6 +470,82 @@ const GEMINI_2_5_FLASH_TTS = {
|
|
|
472
470
|
GeminiCachedContentOptions
|
|
473
471
|
>
|
|
474
472
|
|
|
473
|
+
/**
|
|
474
|
+
* Gemini 3.1 Flash TTS Preview - latest expressive TTS model with
|
|
475
|
+
* 200+ audio tags, 70+ languages, and multi-speaker dialogue support.
|
|
476
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-tts-preview
|
|
477
|
+
*/
|
|
478
|
+
const GEMINI_3_1_FLASH_TTS = {
|
|
479
|
+
name: 'gemini-3.1-flash-tts-preview',
|
|
480
|
+
max_input_tokens: 32_768,
|
|
481
|
+
max_output_tokens: 16_384,
|
|
482
|
+
knowledge_cutoff: '2025-05-01',
|
|
483
|
+
supports: {
|
|
484
|
+
input: ['text'],
|
|
485
|
+
output: ['audio'],
|
|
486
|
+
capabilities: ['audio_generation', 'batch_api'],
|
|
487
|
+
},
|
|
488
|
+
pricing: {
|
|
489
|
+
input: {
|
|
490
|
+
normal: 0.5,
|
|
491
|
+
},
|
|
492
|
+
output: {
|
|
493
|
+
normal: 10,
|
|
494
|
+
},
|
|
495
|
+
},
|
|
496
|
+
} as const satisfies ModelMeta<
|
|
497
|
+
GeminiToolConfigOptions &
|
|
498
|
+
GeminiSafetyOptions &
|
|
499
|
+
GeminiCommonConfigOptions &
|
|
500
|
+
GeminiCachedContentOptions
|
|
501
|
+
>
|
|
502
|
+
|
|
503
|
+
/**
|
|
504
|
+
* Lyria 3 Pro Preview — Google's flagship music generation model.
|
|
505
|
+
* Generates full-length songs with multiple verses, choruses, and bridges.
|
|
506
|
+
* Outputs MP3 or WAV at 48 kHz stereo.
|
|
507
|
+
* @see https://ai.google.dev/gemini-api/docs/models/lyria-3-pro-preview
|
|
508
|
+
*/
|
|
509
|
+
const LYRIA_3_PRO = {
|
|
510
|
+
name: 'lyria-3-pro-preview',
|
|
511
|
+
max_input_tokens: 131_072,
|
|
512
|
+
supports: {
|
|
513
|
+
input: ['text', 'image'],
|
|
514
|
+
output: ['audio'],
|
|
515
|
+
capabilities: ['audio_generation'],
|
|
516
|
+
},
|
|
517
|
+
pricing: {
|
|
518
|
+
input: {
|
|
519
|
+
normal: 0,
|
|
520
|
+
},
|
|
521
|
+
output: {
|
|
522
|
+
normal: 0,
|
|
523
|
+
},
|
|
524
|
+
},
|
|
525
|
+
} as const satisfies ModelMeta
|
|
526
|
+
|
|
527
|
+
/**
|
|
528
|
+
* Lyria 3 Clip Preview — 30-second music clips in MP3 format.
|
|
529
|
+
* @see https://ai.google.dev/gemini-api/docs/music-generation
|
|
530
|
+
*/
|
|
531
|
+
const LYRIA_3_CLIP = {
|
|
532
|
+
name: 'lyria-3-clip-preview',
|
|
533
|
+
max_input_tokens: 131_072,
|
|
534
|
+
supports: {
|
|
535
|
+
input: ['text', 'image'],
|
|
536
|
+
output: ['audio'],
|
|
537
|
+
capabilities: ['audio_generation'],
|
|
538
|
+
},
|
|
539
|
+
pricing: {
|
|
540
|
+
input: {
|
|
541
|
+
normal: 0,
|
|
542
|
+
},
|
|
543
|
+
output: {
|
|
544
|
+
normal: 0,
|
|
545
|
+
},
|
|
546
|
+
},
|
|
547
|
+
} as const satisfies ModelMeta
|
|
548
|
+
|
|
475
549
|
const GEMINI_2_5_FLASH_LITE = {
|
|
476
550
|
name: 'gemini-2.5-flash-lite',
|
|
477
551
|
max_input_tokens: 1_048_576,
|
|
@@ -580,7 +654,7 @@ const GEMINI_2_FLASH_IMAGE = {
|
|
|
580
654
|
knowledge_cutoff: '2024-08-01',
|
|
581
655
|
supports: {
|
|
582
656
|
input: ['text', 'image', 'audio', 'video'],
|
|
583
|
-
output: ['text'],
|
|
657
|
+
output: ['text', 'image'],
|
|
584
658
|
capabilities: ['batch_api', 'caching', 'structured_output'],
|
|
585
659
|
tools: [],
|
|
586
660
|
},
|
|
@@ -930,10 +1004,20 @@ export const GEMINI_IMAGE_MODELS = [
|
|
|
930
1004
|
* @experimental Gemini TTS is an experimental feature and may change.
|
|
931
1005
|
*/
|
|
932
1006
|
export const GEMINI_TTS_MODELS = [
|
|
1007
|
+
GEMINI_3_1_FLASH_TTS.name,
|
|
933
1008
|
GEMINI_2_5_FLASH_TTS.name,
|
|
934
1009
|
GEMINI_2_5_PRO_TTS.name,
|
|
935
1010
|
] as const
|
|
936
1011
|
|
|
1012
|
+
/**
|
|
1013
|
+
* Audio generation models (Lyria music generation).
|
|
1014
|
+
* @experimental Lyria music generation is an experimental feature and may change.
|
|
1015
|
+
*/
|
|
1016
|
+
export const GEMINI_AUDIO_MODELS = [
|
|
1017
|
+
LYRIA_3_PRO.name,
|
|
1018
|
+
LYRIA_3_CLIP.name,
|
|
1019
|
+
] as const
|
|
1020
|
+
|
|
937
1021
|
/**
|
|
938
1022
|
* Available voice names for Gemini TTS
|
|
939
1023
|
* @see https://ai.google.dev/gemini-api/docs/speech-generation
|