@tanstack/ai-gemini 0.8.9 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.js +28 -13
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/summarize.js +98 -70
- package/dist/esm/adapters/summarize.js.map +1 -1
- package/dist/esm/adapters/text.d.ts +9 -4
- package/dist/esm/adapters/text.js +21 -2
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/tts.js +51 -38
- package/dist/esm/adapters/tts.js.map +1 -1
- package/dist/esm/index.d.ts +1 -1
- package/dist/esm/model-meta.d.ts +38 -10
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/tools/code-execution-tool.d.ts +6 -3
- package/dist/esm/tools/code-execution-tool.js +8 -0
- package/dist/esm/tools/code-execution-tool.js.map +1 -1
- package/dist/esm/tools/computer-use-tool.d.ts +6 -3
- package/dist/esm/tools/computer-use-tool.js +11 -0
- package/dist/esm/tools/computer-use-tool.js.map +1 -1
- package/dist/esm/tools/file-search-tool.d.ts +6 -3
- package/dist/esm/tools/file-search-tool.js +9 -1
- package/dist/esm/tools/file-search-tool.js.map +1 -1
- package/dist/esm/tools/function-declaration-tool.js +29 -0
- package/dist/esm/tools/function-declaration-tool.js.map +1 -0
- package/dist/esm/tools/google-maps-tool.d.ts +6 -3
- package/dist/esm/tools/google-maps-tool.js +9 -1
- package/dist/esm/tools/google-maps-tool.js.map +1 -1
- package/dist/esm/tools/google-search-retriveal-tool.d.ts +6 -3
- package/dist/esm/tools/google-search-retriveal-tool.js +9 -1
- package/dist/esm/tools/google-search-retriveal-tool.js.map +1 -1
- package/dist/esm/tools/google-search-tool.d.ts +6 -3
- package/dist/esm/tools/google-search-tool.js +9 -1
- package/dist/esm/tools/google-search-tool.js.map +1 -1
- package/dist/esm/tools/index.d.ts +9 -0
- package/dist/esm/tools/index.js +21 -0
- package/dist/esm/tools/index.js.map +1 -0
- package/dist/esm/tools/url-context-tool.d.ts +6 -3
- package/dist/esm/tools/url-context-tool.js +9 -1
- package/dist/esm/tools/url-context-tool.js.map +1 -1
- package/package.json +8 -3
- package/src/adapters/image.ts +31 -15
- package/src/adapters/summarize.ts +111 -80
- package/src/adapters/text.ts +41 -5
- package/src/adapters/tts.ts +62 -48
- package/src/index.ts +1 -0
- package/src/model-meta.ts +59 -72
- package/src/tools/code-execution-tool.ts +10 -4
- package/src/tools/computer-use-tool.ts +13 -5
- package/src/tools/file-search-tool.ts +13 -5
- package/src/tools/google-maps-tool.ts +13 -5
- package/src/tools/google-search-retriveal-tool.ts +15 -6
- package/src/tools/google-search-tool.ts +13 -5
- package/src/tools/index.ts +51 -0
- package/src/tools/url-context-tool.ts +10 -4
package/src/adapters/image.ts
CHANGED
|
@@ -81,27 +81,43 @@ export class GeminiImageAdapter<
|
|
|
81
81
|
async generateImages(
|
|
82
82
|
options: ImageGenerationOptions<GeminiImageProviderOptions>,
|
|
83
83
|
): Promise<ImageGenerationResult> {
|
|
84
|
-
const { model, prompt } = options
|
|
84
|
+
const { model, prompt, logger } = options
|
|
85
85
|
|
|
86
|
-
|
|
86
|
+
logger.request(
|
|
87
|
+
`activity=generateImage provider=gemini model=${this.model}`,
|
|
88
|
+
{
|
|
89
|
+
provider: 'gemini',
|
|
90
|
+
model: this.model,
|
|
91
|
+
},
|
|
92
|
+
)
|
|
87
93
|
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
}
|
|
94
|
+
try {
|
|
95
|
+
validatePrompt({ prompt, model })
|
|
91
96
|
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
97
|
+
if (this.isGeminiImageModel(model)) {
|
|
98
|
+
return await this.generateWithGeminiApi(options)
|
|
99
|
+
}
|
|
95
100
|
|
|
96
|
-
|
|
101
|
+
// Imagen models path (generateImages API)
|
|
102
|
+
validateImageSize(model, options.size)
|
|
103
|
+
validateNumberOfImages(model, options.numberOfImages)
|
|
97
104
|
|
|
98
|
-
|
|
99
|
-
model,
|
|
100
|
-
prompt,
|
|
101
|
-
config,
|
|
102
|
-
})
|
|
105
|
+
const config = this.buildImagenConfig(options)
|
|
103
106
|
|
|
104
|
-
|
|
107
|
+
const response = await this.client.models.generateImages({
|
|
108
|
+
model,
|
|
109
|
+
prompt,
|
|
110
|
+
config,
|
|
111
|
+
})
|
|
112
|
+
|
|
113
|
+
return this.transformImagenResponse(model, response)
|
|
114
|
+
} catch (error) {
|
|
115
|
+
logger.errors('gemini.generateImage fatal', {
|
|
116
|
+
error,
|
|
117
|
+
source: 'gemini.generateImage',
|
|
118
|
+
})
|
|
119
|
+
throw error
|
|
120
|
+
}
|
|
105
121
|
}
|
|
106
122
|
|
|
107
123
|
private isGeminiImageModel(model: string): boolean {
|
|
@@ -81,8 +81,14 @@ export class GeminiSummarizeAdapter<
|
|
|
81
81
|
}
|
|
82
82
|
|
|
83
83
|
async summarize(options: SummarizationOptions): Promise<SummarizationResult> {
|
|
84
|
+
const { logger } = options
|
|
84
85
|
const model = options.model
|
|
85
86
|
|
|
87
|
+
logger.request(`activity=summarize provider=gemini`, {
|
|
88
|
+
provider: 'gemini',
|
|
89
|
+
model,
|
|
90
|
+
})
|
|
91
|
+
|
|
86
92
|
// Build the system prompt based on format
|
|
87
93
|
const formatInstructions = this.getFormatInstructions(options.style)
|
|
88
94
|
const lengthInstructions = options.maxLength
|
|
@@ -91,40 +97,49 @@ export class GeminiSummarizeAdapter<
|
|
|
91
97
|
|
|
92
98
|
const systemPrompt = `You are a helpful assistant that summarizes text. ${formatInstructions}${lengthInstructions}`
|
|
93
99
|
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
100
|
+
try {
|
|
101
|
+
const response = await this.client.models.generateContent({
|
|
102
|
+
model,
|
|
103
|
+
contents: [
|
|
104
|
+
{
|
|
105
|
+
role: 'user',
|
|
106
|
+
parts: [
|
|
107
|
+
{ text: `Please summarize the following:\n\n${options.text}` },
|
|
108
|
+
],
|
|
109
|
+
},
|
|
110
|
+
],
|
|
111
|
+
config: {
|
|
112
|
+
systemInstruction: systemPrompt,
|
|
102
113
|
},
|
|
103
|
-
|
|
104
|
-
config: {
|
|
105
|
-
systemInstruction: systemPrompt,
|
|
106
|
-
},
|
|
107
|
-
})
|
|
114
|
+
})
|
|
108
115
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
116
|
+
const summary = response.text ?? ''
|
|
117
|
+
const inputTokens = response.usageMetadata?.promptTokenCount ?? 0
|
|
118
|
+
const outputTokens = response.usageMetadata?.candidatesTokenCount ?? 0
|
|
112
119
|
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
120
|
+
return {
|
|
121
|
+
id: generateId('sum'),
|
|
122
|
+
model,
|
|
123
|
+
summary,
|
|
124
|
+
usage: {
|
|
125
|
+
promptTokens: inputTokens,
|
|
126
|
+
completionTokens: outputTokens,
|
|
127
|
+
totalTokens: inputTokens + outputTokens,
|
|
128
|
+
},
|
|
129
|
+
}
|
|
130
|
+
} catch (error) {
|
|
131
|
+
logger.errors('gemini.summarize fatal', {
|
|
132
|
+
error,
|
|
133
|
+
source: 'gemini.summarize',
|
|
134
|
+
})
|
|
135
|
+
throw error
|
|
122
136
|
}
|
|
123
137
|
}
|
|
124
138
|
|
|
125
139
|
async *summarizeStream(
|
|
126
140
|
options: SummarizationOptions,
|
|
127
141
|
): AsyncIterable<StreamChunk> {
|
|
142
|
+
const { logger } = options
|
|
128
143
|
const model = options.model
|
|
129
144
|
const id = generateId('sum')
|
|
130
145
|
let accumulatedContent = ''
|
|
@@ -139,69 +154,85 @@ export class GeminiSummarizeAdapter<
|
|
|
139
154
|
|
|
140
155
|
const systemPrompt = `You are a helpful assistant that summarizes text. ${formatInstructions}${lengthInstructions}`
|
|
141
156
|
|
|
142
|
-
|
|
157
|
+
logger.request(`activity=summarize provider=gemini`, {
|
|
158
|
+
provider: 'gemini',
|
|
143
159
|
model,
|
|
144
|
-
|
|
145
|
-
{
|
|
146
|
-
role: 'user',
|
|
147
|
-
parts: [
|
|
148
|
-
{ text: `Please summarize the following:\n\n${options.text}` },
|
|
149
|
-
],
|
|
150
|
-
},
|
|
151
|
-
],
|
|
152
|
-
config: {
|
|
153
|
-
systemInstruction: systemPrompt,
|
|
154
|
-
},
|
|
160
|
+
stream: true,
|
|
155
161
|
})
|
|
156
162
|
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
+
try {
|
|
164
|
+
const result = await this.client.models.generateContentStream({
|
|
165
|
+
model,
|
|
166
|
+
contents: [
|
|
167
|
+
{
|
|
168
|
+
role: 'user',
|
|
169
|
+
parts: [
|
|
170
|
+
{ text: `Please summarize the following:\n\n${options.text}` },
|
|
171
|
+
],
|
|
172
|
+
},
|
|
173
|
+
],
|
|
174
|
+
config: {
|
|
175
|
+
systemInstruction: systemPrompt,
|
|
176
|
+
},
|
|
177
|
+
})
|
|
163
178
|
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
179
|
+
for await (const chunk of result) {
|
|
180
|
+
logger.provider(`provider=gemini`, { chunk })
|
|
181
|
+
// Track usage metadata
|
|
182
|
+
if (chunk.usageMetadata) {
|
|
183
|
+
inputTokens = chunk.usageMetadata.promptTokenCount ?? inputTokens
|
|
184
|
+
outputTokens =
|
|
185
|
+
chunk.usageMetadata.candidatesTokenCount ?? outputTokens
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
if (chunk.candidates?.[0]?.content?.parts) {
|
|
189
|
+
for (const part of chunk.candidates[0].content.parts) {
|
|
190
|
+
if (part.text) {
|
|
191
|
+
accumulatedContent += part.text
|
|
192
|
+
yield asChunk({
|
|
193
|
+
type: 'TEXT_MESSAGE_CONTENT',
|
|
194
|
+
messageId: id,
|
|
195
|
+
model,
|
|
196
|
+
timestamp: Date.now(),
|
|
197
|
+
delta: part.text,
|
|
198
|
+
content: accumulatedContent,
|
|
199
|
+
})
|
|
200
|
+
}
|
|
176
201
|
}
|
|
177
202
|
}
|
|
178
|
-
}
|
|
179
203
|
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
+
// Check for finish reason
|
|
205
|
+
const finishReason = chunk.candidates?.[0]?.finishReason
|
|
206
|
+
if (
|
|
207
|
+
finishReason === FinishReason.STOP ||
|
|
208
|
+
finishReason === FinishReason.MAX_TOKENS ||
|
|
209
|
+
finishReason === FinishReason.SAFETY
|
|
210
|
+
) {
|
|
211
|
+
yield asChunk({
|
|
212
|
+
type: 'RUN_FINISHED',
|
|
213
|
+
runId: id,
|
|
214
|
+
model,
|
|
215
|
+
timestamp: Date.now(),
|
|
216
|
+
finishReason:
|
|
217
|
+
finishReason === FinishReason.STOP
|
|
218
|
+
? 'stop'
|
|
219
|
+
: finishReason === FinishReason.MAX_TOKENS
|
|
220
|
+
? 'length'
|
|
221
|
+
: 'content_filter',
|
|
222
|
+
usage: {
|
|
223
|
+
promptTokens: inputTokens,
|
|
224
|
+
completionTokens: outputTokens,
|
|
225
|
+
totalTokens: inputTokens + outputTokens,
|
|
226
|
+
},
|
|
227
|
+
})
|
|
228
|
+
}
|
|
204
229
|
}
|
|
230
|
+
} catch (error) {
|
|
231
|
+
logger.errors('gemini.summarize fatal', {
|
|
232
|
+
error,
|
|
233
|
+
source: 'gemini.summarize',
|
|
234
|
+
})
|
|
235
|
+
throw error
|
|
205
236
|
}
|
|
206
237
|
}
|
|
207
238
|
|
package/src/adapters/text.ts
CHANGED
|
@@ -9,12 +9,14 @@ import {
|
|
|
9
9
|
import type {
|
|
10
10
|
GEMINI_MODELS,
|
|
11
11
|
GeminiChatModelProviderOptionsByName,
|
|
12
|
+
GeminiChatModelToolCapabilitiesByName,
|
|
12
13
|
GeminiModelInputModalitiesByName,
|
|
13
14
|
} from '../model-meta'
|
|
14
15
|
import type {
|
|
15
16
|
StructuredOutputOptions,
|
|
16
17
|
StructuredOutputResult,
|
|
17
18
|
} from '@tanstack/ai/adapters'
|
|
19
|
+
import type { InternalLogger } from '@tanstack/ai/adapter-internals'
|
|
18
20
|
import type {
|
|
19
21
|
Content,
|
|
20
22
|
GenerateContentParameters,
|
|
@@ -71,6 +73,15 @@ type ResolveInputModalities<TModel extends string> =
|
|
|
71
73
|
? GeminiModelInputModalitiesByName[TModel]
|
|
72
74
|
: readonly ['text', 'image', 'audio', 'video', 'document']
|
|
73
75
|
|
|
76
|
+
/**
|
|
77
|
+
* Resolve tool capabilities for a specific model.
|
|
78
|
+
* If the model has explicit tools in the map, use those; otherwise use empty tuple.
|
|
79
|
+
*/
|
|
80
|
+
type ResolveToolCapabilities<TModel extends string> =
|
|
81
|
+
TModel extends keyof GeminiChatModelToolCapabilitiesByName
|
|
82
|
+
? NonNullable<GeminiChatModelToolCapabilitiesByName[TModel]>
|
|
83
|
+
: readonly []
|
|
84
|
+
|
|
74
85
|
// ===========================
|
|
75
86
|
// Adapter Implementation
|
|
76
87
|
// ===========================
|
|
@@ -83,14 +94,17 @@ type ResolveInputModalities<TModel extends string> =
|
|
|
83
94
|
*/
|
|
84
95
|
export class GeminiTextAdapter<
|
|
85
96
|
TModel extends (typeof GEMINI_MODELS)[number],
|
|
86
|
-
TProviderOptions extends
|
|
97
|
+
TProviderOptions extends Record<string, any> = ResolveProviderOptions<TModel>,
|
|
87
98
|
TInputModalities extends ReadonlyArray<Modality> =
|
|
88
99
|
ResolveInputModalities<TModel>,
|
|
100
|
+
TToolCapabilities extends ReadonlyArray<string> =
|
|
101
|
+
ResolveToolCapabilities<TModel>,
|
|
89
102
|
> extends BaseTextAdapter<
|
|
90
103
|
TModel,
|
|
91
104
|
TProviderOptions,
|
|
92
105
|
TInputModalities,
|
|
93
|
-
GeminiMessageMetadataByModality
|
|
106
|
+
GeminiMessageMetadataByModality,
|
|
107
|
+
TToolCapabilities
|
|
94
108
|
> {
|
|
95
109
|
readonly kind = 'text' as const
|
|
96
110
|
readonly name = 'gemini' as const
|
|
@@ -106,14 +120,23 @@ export class GeminiTextAdapter<
|
|
|
106
120
|
options: TextOptions<GeminiTextProviderOptions>,
|
|
107
121
|
): AsyncIterable<StreamChunk> {
|
|
108
122
|
const mappedOptions = this.mapCommonOptionsToGemini(options)
|
|
123
|
+
const { logger } = options
|
|
109
124
|
|
|
110
125
|
try {
|
|
126
|
+
logger.request(
|
|
127
|
+
`activity=chat provider=gemini model=${this.model} messages=${options.messages.length} tools=${options.tools?.length ?? 0} stream=true`,
|
|
128
|
+
{ provider: 'gemini', model: this.model },
|
|
129
|
+
)
|
|
111
130
|
const result =
|
|
112
131
|
await this.client.models.generateContentStream(mappedOptions)
|
|
113
132
|
|
|
114
|
-
yield* this.processStreamChunks(result, options)
|
|
133
|
+
yield* this.processStreamChunks(result, options, logger)
|
|
115
134
|
} catch (error) {
|
|
116
135
|
const timestamp = Date.now()
|
|
136
|
+
logger.errors('gemini.chatStream fatal', {
|
|
137
|
+
error,
|
|
138
|
+
source: 'gemini.chatStream',
|
|
139
|
+
})
|
|
117
140
|
yield asChunk({
|
|
118
141
|
type: 'RUN_ERROR',
|
|
119
142
|
model: options.model,
|
|
@@ -141,10 +164,15 @@ export class GeminiTextAdapter<
|
|
|
141
164
|
options: StructuredOutputOptions<GeminiTextProviderOptions>,
|
|
142
165
|
): Promise<StructuredOutputResult<unknown>> {
|
|
143
166
|
const { chatOptions, outputSchema } = options
|
|
167
|
+
const { logger } = chatOptions
|
|
144
168
|
|
|
145
169
|
const mappedOptions = this.mapCommonOptionsToGemini(chatOptions)
|
|
146
170
|
|
|
147
171
|
try {
|
|
172
|
+
logger.request(
|
|
173
|
+
`activity=chat provider=gemini model=${this.model} messages=${chatOptions.messages.length} tools=${chatOptions.tools?.length ?? 0} stream=false`,
|
|
174
|
+
{ provider: 'gemini', model: this.model },
|
|
175
|
+
)
|
|
148
176
|
// Add structured output configuration
|
|
149
177
|
const result = await this.client.models.generateContent({
|
|
150
178
|
...mappedOptions,
|
|
@@ -173,6 +201,10 @@ export class GeminiTextAdapter<
|
|
|
173
201
|
rawText,
|
|
174
202
|
}
|
|
175
203
|
} catch (error) {
|
|
204
|
+
logger.errors('gemini.structuredOutput fatal', {
|
|
205
|
+
error,
|
|
206
|
+
source: 'gemini.structuredOutput',
|
|
207
|
+
})
|
|
176
208
|
throw new Error(
|
|
177
209
|
error instanceof Error
|
|
178
210
|
? error.message
|
|
@@ -201,6 +233,7 @@ export class GeminiTextAdapter<
|
|
|
201
233
|
private async *processStreamChunks(
|
|
202
234
|
result: AsyncGenerator<GenerateContentResponse, unknown, unknown>,
|
|
203
235
|
options: TextOptions<GeminiTextProviderOptions>,
|
|
236
|
+
logger: InternalLogger,
|
|
204
237
|
): AsyncIterable<StreamChunk> {
|
|
205
238
|
const model = options.model
|
|
206
239
|
const timestamp = Date.now()
|
|
@@ -230,6 +263,7 @@ export class GeminiTextAdapter<
|
|
|
230
263
|
let hasEmittedStepStarted = false
|
|
231
264
|
|
|
232
265
|
for await (const chunk of result) {
|
|
266
|
+
logger.provider(`provider=gemini`, { chunk })
|
|
233
267
|
// Emit RUN_STARTED on first chunk
|
|
234
268
|
if (!hasEmittedRunStarted) {
|
|
235
269
|
hasEmittedRunStarted = true
|
|
@@ -810,7 +844,8 @@ export function createGeminiChat<TModel extends (typeof GEMINI_MODELS)[number]>(
|
|
|
810
844
|
): GeminiTextAdapter<
|
|
811
845
|
TModel,
|
|
812
846
|
ResolveProviderOptions<TModel>,
|
|
813
|
-
ResolveInputModalities<TModel
|
|
847
|
+
ResolveInputModalities<TModel>,
|
|
848
|
+
ResolveToolCapabilities<TModel>
|
|
814
849
|
> {
|
|
815
850
|
return new GeminiTextAdapter({ apiKey, ...config }, model)
|
|
816
851
|
}
|
|
@@ -825,7 +860,8 @@ export function geminiText<TModel extends (typeof GEMINI_MODELS)[number]>(
|
|
|
825
860
|
): GeminiTextAdapter<
|
|
826
861
|
TModel,
|
|
827
862
|
ResolveProviderOptions<TModel>,
|
|
828
|
-
ResolveInputModalities<TModel
|
|
863
|
+
ResolveInputModalities<TModel>,
|
|
864
|
+
ResolveToolCapabilities<TModel>
|
|
829
865
|
> {
|
|
830
866
|
const apiKey = getGeminiApiKeyFromEnv()
|
|
831
867
|
return createGeminiChat(model, apiKey, config)
|
package/src/adapters/tts.ts
CHANGED
|
@@ -98,63 +98,77 @@ export class GeminiTTSAdapter<
|
|
|
98
98
|
async generateSpeech(
|
|
99
99
|
options: TTSOptions<GeminiTTSProviderOptions>,
|
|
100
100
|
): Promise<TTSResult> {
|
|
101
|
+
const { logger } = options
|
|
101
102
|
const { model, text, modelOptions } = options
|
|
102
103
|
|
|
104
|
+
logger.request(`activity=generateSpeech provider=gemini model=${model}`, {
|
|
105
|
+
provider: 'gemini',
|
|
106
|
+
model,
|
|
107
|
+
})
|
|
108
|
+
|
|
103
109
|
const voiceConfig = modelOptions?.voiceConfig || {
|
|
104
110
|
prebuiltVoiceConfig: {
|
|
105
111
|
voiceName: 'Kore',
|
|
106
112
|
},
|
|
107
113
|
}
|
|
108
114
|
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
+
try {
|
|
116
|
+
const response = await this.client.models.generateContent({
|
|
117
|
+
model,
|
|
118
|
+
contents: [
|
|
119
|
+
{
|
|
120
|
+
role: 'user',
|
|
121
|
+
parts: [{ text }],
|
|
122
|
+
},
|
|
123
|
+
],
|
|
124
|
+
config: {
|
|
125
|
+
responseModalities: ['AUDIO'],
|
|
126
|
+
speechConfig: {
|
|
127
|
+
voiceConfig,
|
|
128
|
+
...(modelOptions?.languageCode && {
|
|
129
|
+
languageCode: modelOptions.languageCode,
|
|
130
|
+
}),
|
|
131
|
+
},
|
|
115
132
|
},
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
audio: audioBase64,
|
|
156
|
-
format,
|
|
157
|
-
contentType: mimeType,
|
|
133
|
+
...(modelOptions?.systemInstruction && {
|
|
134
|
+
systemInstruction: modelOptions.systemInstruction,
|
|
135
|
+
}),
|
|
136
|
+
})
|
|
137
|
+
|
|
138
|
+
// Extract audio data from response
|
|
139
|
+
const candidate = response.candidates?.[0]
|
|
140
|
+
const parts = candidate?.content?.parts
|
|
141
|
+
|
|
142
|
+
if (!parts || parts.length === 0) {
|
|
143
|
+
throw new Error('No audio output received from Gemini TTS')
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
// Look for inline data (audio)
|
|
147
|
+
const audioPart = parts.find((part: any) =>
|
|
148
|
+
part.inlineData?.mimeType?.startsWith('audio/'),
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
if (!audioPart || !audioPart.inlineData || !audioPart.inlineData.data) {
|
|
152
|
+
throw new Error('No audio data in Gemini TTS response')
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const audioBase64 = audioPart.inlineData.data
|
|
156
|
+
const mimeType = audioPart.inlineData.mimeType || 'audio/wav'
|
|
157
|
+
const format = mimeType.split('/')[1] || 'wav'
|
|
158
|
+
|
|
159
|
+
return {
|
|
160
|
+
id: generateId(this.name),
|
|
161
|
+
model,
|
|
162
|
+
audio: audioBase64,
|
|
163
|
+
format,
|
|
164
|
+
contentType: mimeType,
|
|
165
|
+
}
|
|
166
|
+
} catch (error) {
|
|
167
|
+
logger.errors('gemini.generateSpeech fatal', {
|
|
168
|
+
error,
|
|
169
|
+
source: 'gemini.generateSpeech',
|
|
170
|
+
})
|
|
171
|
+
throw error
|
|
158
172
|
}
|
|
159
173
|
}
|
|
160
174
|
}
|