@tanstack/ai-gemini 0.0.3 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +26 -0
  2. package/dist/esm/adapters/image.d.ts +83 -0
  3. package/dist/esm/adapters/image.js +60 -0
  4. package/dist/esm/adapters/image.js.map +1 -0
  5. package/dist/esm/adapters/summarize.d.ts +53 -0
  6. package/dist/esm/adapters/summarize.js +139 -0
  7. package/dist/esm/adapters/summarize.js.map +1 -0
  8. package/dist/esm/adapters/text.d.ts +63 -0
  9. package/dist/esm/{gemini-adapter.js → adapters/text.js} +86 -132
  10. package/dist/esm/adapters/text.js.map +1 -0
  11. package/dist/esm/adapters/tts.d.ts +129 -0
  12. package/dist/esm/adapters/tts.js +78 -0
  13. package/dist/esm/adapters/tts.js.map +1 -0
  14. package/dist/esm/image/image-provider-options.d.ts +139 -0
  15. package/dist/esm/image/image-provider-options.js +53 -0
  16. package/dist/esm/image/image-provider-options.js.map +1 -0
  17. package/dist/esm/index.d.ts +15 -2
  18. package/dist/esm/index.js +22 -4
  19. package/dist/esm/index.js.map +1 -1
  20. package/dist/esm/model-meta.d.ts +128 -1
  21. package/dist/esm/model-meta.js +71 -5
  22. package/dist/esm/model-meta.js.map +1 -1
  23. package/dist/esm/tools/tool-converter.js +5 -3
  24. package/dist/esm/tools/tool-converter.js.map +1 -1
  25. package/dist/esm/utils/client.d.ts +17 -0
  26. package/dist/esm/utils/client.js +25 -0
  27. package/dist/esm/utils/client.js.map +1 -0
  28. package/dist/esm/utils/index.d.ts +1 -0
  29. package/package.json +5 -4
  30. package/src/adapters/image.ts +188 -0
  31. package/src/adapters/summarize.ts +242 -0
  32. package/src/{gemini-adapter.ts → adapters/text.ts} +196 -245
  33. package/src/adapters/tts.ts +223 -0
  34. package/src/image/image-provider-options.ts +239 -0
  35. package/src/index.ts +66 -2
  36. package/src/model-meta.ts +87 -62
  37. package/src/tools/tool-converter.ts +6 -5
  38. package/src/utils/client.ts +43 -0
  39. package/src/utils/index.ts +6 -0
  40. package/dist/esm/gemini-adapter.d.ts +0 -71
  41. package/dist/esm/gemini-adapter.js.map +0 -1
@@ -1,75 +1,109 @@
1
- import { FinishReason, GoogleGenAI } from '@google/genai'
2
- import { BaseAdapter } from '@tanstack/ai'
3
- import { GEMINI_EMBEDDING_MODELS, GEMINI_MODELS } from './model-meta'
4
- import { convertToolsToProviderFormat } from './tools/tool-converter'
5
- import type {
6
- AIAdapterConfig,
7
- ChatOptions,
8
- ContentPart,
9
- EmbeddingOptions,
10
- EmbeddingResult,
11
- ModelMessage,
12
- StreamChunk,
13
- SummarizationOptions,
14
- SummarizationResult,
15
- } from '@tanstack/ai'
1
+ import { FinishReason } from '@google/genai'
2
+ import { BaseTextAdapter } from '@tanstack/ai/adapters'
3
+ import { convertToolsToProviderFormat } from '../tools/tool-converter'
4
+ import {
5
+ createGeminiClient,
6
+ generateId,
7
+ getGeminiApiKeyFromEnv,
8
+ } from '../utils'
16
9
  import type {
10
+ GEMINI_MODELS,
17
11
  GeminiChatModelProviderOptionsByName,
18
12
  GeminiModelInputModalitiesByName,
19
- } from './model-meta'
20
- import type { ExternalTextProviderOptions } from './text/text-provider-options'
13
+ } from '../model-meta'
14
+ import type {
15
+ StructuredOutputOptions,
16
+ StructuredOutputResult,
17
+ } from '@tanstack/ai/adapters'
21
18
  import type {
22
19
  GenerateContentParameters,
23
20
  GenerateContentResponse,
21
+ GoogleGenAI,
24
22
  Part,
25
23
  } from '@google/genai'
24
+ import type {
25
+ ContentPart,
26
+ Modality,
27
+ ModelMessage,
28
+ StreamChunk,
29
+ TextOptions,
30
+ } from '@tanstack/ai'
31
+ import type { ExternalTextProviderOptions } from '../text/text-provider-options'
26
32
  import type {
27
33
  GeminiAudioMetadata,
28
34
  GeminiDocumentMetadata,
29
35
  GeminiImageMetadata,
30
36
  GeminiMessageMetadataByModality,
31
37
  GeminiVideoMetadata,
32
- } from './message-types'
38
+ } from '../message-types'
39
+ import type { GeminiClientConfig } from '../utils'
33
40
 
34
- export interface GeminiAdapterConfig extends AIAdapterConfig {
35
- apiKey: string
36
- }
41
+ /**
42
+ * Configuration for Gemini text adapter
43
+ */
44
+ export interface GeminiTextConfig extends GeminiClientConfig {}
37
45
 
38
46
  /**
39
- * Gemini-specific provider options
40
- * Based on Google Generative AI SDK
41
- * @see https://ai.google.dev/api/rest/v1/GenerationConfig
47
+ * Gemini-specific provider options for text/chat
42
48
  */
43
- export type GeminiProviderOptions = ExternalTextProviderOptions
49
+ export type GeminiTextProviderOptions = ExternalTextProviderOptions
44
50
 
45
- export class GeminiAdapter extends BaseAdapter<
46
- typeof GEMINI_MODELS,
47
- typeof GEMINI_EMBEDDING_MODELS,
48
- GeminiProviderOptions,
49
- Record<string, any>,
50
- GeminiChatModelProviderOptionsByName,
51
- GeminiModelInputModalitiesByName,
51
+ // ===========================
52
+ // Type Resolution Helpers
53
+ // ===========================
54
+
55
+ /**
56
+ * Resolve provider options for a specific model.
57
+ * If the model has explicit options in the map, use those; otherwise use base options.
58
+ */
59
+ type ResolveProviderOptions<TModel extends string> =
60
+ TModel extends keyof GeminiChatModelProviderOptionsByName
61
+ ? GeminiChatModelProviderOptionsByName[TModel]
62
+ : GeminiTextProviderOptions
63
+
64
+ /**
65
+ * Resolve input modalities for a specific model.
66
+ * If the model has explicit modalities in the map, use those; otherwise use all modalities.
67
+ */
68
+ type ResolveInputModalities<TModel extends string> =
69
+ TModel extends keyof GeminiModelInputModalitiesByName
70
+ ? GeminiModelInputModalitiesByName[TModel]
71
+ : readonly ['text', 'image', 'audio', 'video', 'document']
72
+
73
+ // ===========================
74
+ // Adapter Implementation
75
+ // ===========================
76
+
77
+ /**
78
+ * Gemini Text (Chat) Adapter
79
+ *
80
+ * Tree-shakeable adapter for Gemini chat/text completion functionality.
81
+ * Import only what you need for smaller bundle sizes.
82
+ */
83
+ export class GeminiTextAdapter<
84
+ TModel extends (typeof GEMINI_MODELS)[number],
85
+ TProviderOptions extends object = ResolveProviderOptions<TModel>,
86
+ TInputModalities extends ReadonlyArray<Modality> =
87
+ ResolveInputModalities<TModel>,
88
+ > extends BaseTextAdapter<
89
+ TModel,
90
+ TProviderOptions,
91
+ TInputModalities,
52
92
  GeminiMessageMetadataByModality
53
93
  > {
54
- name = 'gemini'
55
- models = GEMINI_MODELS
56
- embeddingModels = GEMINI_EMBEDDING_MODELS
57
- declare _modelProviderOptionsByName: GeminiChatModelProviderOptionsByName
58
- declare _modelInputModalitiesByName: GeminiModelInputModalitiesByName
59
- declare _messageMetadataByModality: GeminiMessageMetadataByModality
94
+ readonly kind = 'text' as const
95
+ readonly name = 'gemini' as const
96
+
60
97
  private client: GoogleGenAI
61
98
 
62
- constructor(config: GeminiAdapterConfig) {
63
- super(config)
64
- this.client = new GoogleGenAI({
65
- apiKey: config.apiKey,
66
- })
99
+ constructor(config: GeminiTextConfig, model: TModel) {
100
+ super({}, model)
101
+ this.client = createGeminiClient(config)
67
102
  }
68
103
 
69
104
  async *chatStream(
70
- options: ChatOptions<string, GeminiProviderOptions>,
105
+ options: TextOptions<GeminiTextProviderOptions>,
71
106
  ): AsyncIterable<StreamChunk> {
72
- // Map common options to Gemini format
73
107
  const mappedOptions = this.mapCommonOptionsToGemini(options)
74
108
 
75
109
  try {
@@ -81,7 +115,7 @@ export class GeminiAdapter extends BaseAdapter<
81
115
  const timestamp = Date.now()
82
116
  yield {
83
117
  type: 'error',
84
- id: this.generateId(),
118
+ id: generateId(this.name),
85
119
  model: options.model,
86
120
  timestamp,
87
121
  error: {
@@ -94,125 +128,74 @@ export class GeminiAdapter extends BaseAdapter<
94
128
  }
95
129
  }
96
130
 
97
- async summarize(options: SummarizationOptions): Promise<SummarizationResult> {
98
- const prompt = this.buildSummarizationPrompt(options, options.text)
131
+ /**
132
+ * Generate structured output using Gemini's native JSON response format.
133
+ * Uses responseMimeType: 'application/json' and responseSchema for structured output.
134
+ * The outputSchema is already JSON Schema (converted in the ai layer).
135
+ */
136
+ async structuredOutput(
137
+ options: StructuredOutputOptions<GeminiTextProviderOptions>,
138
+ ): Promise<StructuredOutputResult<unknown>> {
139
+ const { chatOptions, outputSchema } = options
99
140
 
100
- // Use models API like chatCompletion
101
- const result = await this.client.models.generateContent({
102
- model: options.model,
103
- contents: [{ role: 'user', parts: [{ text: prompt }] }],
104
- config: {
105
- temperature: 0.3,
106
- maxOutputTokens: options.maxLength || 500,
107
- },
108
- })
141
+ const mappedOptions = this.mapCommonOptionsToGemini(chatOptions)
109
142
 
110
- // Extract text from candidates or use .text() method
111
- let summary = ''
112
- if (result.candidates?.[0]?.content?.parts) {
113
- const parts = result.candidates[0].content.parts
114
- for (const part of parts) {
115
- if (part.text) {
116
- summary += part.text
117
- }
118
- }
119
- }
143
+ try {
144
+ // Add structured output configuration
145
+ const result = await this.client.models.generateContent({
146
+ ...mappedOptions,
147
+ config: {
148
+ ...mappedOptions.config,
149
+ responseMimeType: 'application/json',
150
+ responseSchema: outputSchema,
151
+ },
152
+ })
120
153
 
121
- if (!summary && typeof result.text === 'string') {
122
- summary = result.text
123
- }
154
+ // Extract text content from the response
155
+ const rawText = this.extractTextFromResponse(result)
124
156
 
125
- const promptTokens = this.estimateTokens(prompt)
126
- const completionTokens = this.estimateTokens(summary)
157
+ // Parse the JSON response
158
+ let parsed: unknown
159
+ try {
160
+ parsed = JSON.parse(rawText)
161
+ } catch {
162
+ throw new Error(
163
+ `Failed to parse structured output as JSON. Content: ${rawText.slice(0, 200)}${rawText.length > 200 ? '...' : ''}`,
164
+ )
165
+ }
127
166
 
128
- return {
129
- id: this.generateId(),
130
- model: options.model,
131
- summary,
132
- usage: {
133
- promptTokens,
134
- completionTokens,
135
- totalTokens: promptTokens + completionTokens,
136
- },
167
+ return {
168
+ data: parsed,
169
+ rawText,
170
+ }
171
+ } catch (error) {
172
+ throw new Error(
173
+ error instanceof Error
174
+ ? error.message
175
+ : 'An unknown error occurred during structured output generation.',
176
+ )
137
177
  }
138
178
  }
139
179
 
140
- async createEmbeddings(options: EmbeddingOptions): Promise<EmbeddingResult> {
141
- const inputs = Array.isArray(options.input)
142
- ? options.input
143
- : [options.input]
144
-
145
- // According to docs: contents can be a string or array of strings
146
- // Response has embeddings (plural) array with values property
147
- const result = await this.client.models.embedContent({
148
- model: options.model,
149
- contents: inputs,
150
- })
180
+ /**
181
+ * Extract text content from a non-streaming response
182
+ */
183
+ private extractTextFromResponse(response: GenerateContentResponse): string {
184
+ let textContent = ''
151
185
 
152
- // Extract embeddings from result.embeddings array
153
- const embeddings: Array<Array<number>> = []
154
- if (result.embeddings && Array.isArray(result.embeddings)) {
155
- for (const embedding of result.embeddings) {
156
- if (embedding.values && Array.isArray(embedding.values)) {
157
- embeddings.push(embedding.values)
158
- } else if (Array.isArray(embedding)) {
159
- embeddings.push(embedding)
186
+ if (response.candidates?.[0]?.content?.parts) {
187
+ for (const part of response.candidates[0].content.parts) {
188
+ if (part.text) {
189
+ textContent += part.text
160
190
  }
161
191
  }
162
192
  }
163
193
 
164
- const promptTokens = inputs.reduce(
165
- (sum, input) => sum + this.estimateTokens(input),
166
- 0,
167
- )
168
-
169
- return {
170
- id: this.generateId(),
171
- model: options.model || 'gemini-embedding-001',
172
- embeddings,
173
- usage: {
174
- promptTokens,
175
- totalTokens: promptTokens,
176
- },
177
- }
178
- }
179
-
180
- private buildSummarizationPrompt(
181
- options: SummarizationOptions,
182
- text: string,
183
- ): string {
184
- let prompt = 'You are a professional summarizer. '
185
-
186
- switch (options.style) {
187
- case 'bullet-points':
188
- prompt += 'Provide a summary in bullet point format. '
189
- break
190
- case 'paragraph':
191
- prompt += 'Provide a summary in paragraph format. '
192
- break
193
- case 'concise':
194
- prompt += 'Provide a very concise summary in 1-2 sentences. '
195
- break
196
- default:
197
- prompt += 'Provide a clear and concise summary. '
198
- }
199
-
200
- if (options.focus && options.focus.length > 0) {
201
- prompt += `Focus on the following aspects: ${options.focus.join(', ')}. `
202
- }
203
-
204
- prompt += `\n\nText to summarize:\n${text}\n\nSummary:`
205
-
206
- return prompt
207
- }
208
-
209
- private estimateTokens(text: string): number {
210
- // Rough approximation: 1 token ≈ 4 characters
211
- return Math.ceil(text.length / 4)
194
+ return textContent
212
195
  }
213
196
 
214
197
  private async *processStreamChunks(
215
- result: AsyncGenerator<GenerateContentResponse, any, any>,
198
+ result: AsyncGenerator<GenerateContentResponse, unknown, unknown>,
216
199
  model: string,
217
200
  ): AsyncIterable<StreamChunk> {
218
201
  const timestamp = Date.now()
@@ -222,20 +205,17 @@ export class GeminiAdapter extends BaseAdapter<
222
205
  { name: string; args: string; index: number }
223
206
  >()
224
207
  let nextToolIndex = 0
225
- // Iterate over the stream result (it's already an AsyncGenerator)
208
+
226
209
  for await (const chunk of result) {
227
- // Extract content from candidates[0].content.parts
228
- // Parts can contain text or functionCall
229
210
  if (chunk.candidates?.[0]?.content?.parts) {
230
211
  const parts = chunk.candidates[0].content.parts
231
212
 
232
213
  for (const part of parts) {
233
- // Handle text content
234
214
  if (part.text) {
235
215
  accumulatedContent += part.text
236
216
  yield {
237
217
  type: 'content',
238
- id: this.generateId(),
218
+ id: generateId(this.name),
239
219
  model,
240
220
  timestamp,
241
221
  delta: part.text,
@@ -244,15 +224,12 @@ export class GeminiAdapter extends BaseAdapter<
244
224
  }
245
225
  }
246
226
 
247
- // Handle function calls (tool calls)
248
- // Check both camelCase (SDK) and snake_case (direct API) formats
249
227
  const functionCall = part.functionCall
250
228
  if (functionCall) {
251
229
  const toolCallId =
252
230
  functionCall.name || `call_${Date.now()}_${nextToolIndex}`
253
231
  const functionArgs = functionCall.args || {}
254
232
 
255
- // Check if we've seen this tool call before (for streaming args)
256
233
  let toolCallData = toolCallMap.get(toolCallId)
257
234
  if (!toolCallData) {
258
235
  toolCallData = {
@@ -265,8 +242,6 @@ export class GeminiAdapter extends BaseAdapter<
265
242
  }
266
243
  toolCallMap.set(toolCallId, toolCallData)
267
244
  } else {
268
- // Merge arguments if streaming
269
-
270
245
  try {
271
246
  const existingArgs = JSON.parse(toolCallData.args)
272
247
  const newArgs =
@@ -276,7 +251,6 @@ export class GeminiAdapter extends BaseAdapter<
276
251
  const mergedArgs = { ...existingArgs, ...newArgs }
277
252
  toolCallData.args = JSON.stringify(mergedArgs)
278
253
  } catch {
279
- // If parsing fails, use new args
280
254
  toolCallData.args =
281
255
  typeof functionArgs === 'string'
282
256
  ? functionArgs
@@ -286,7 +260,7 @@ export class GeminiAdapter extends BaseAdapter<
286
260
 
287
261
  yield {
288
262
  type: 'tool_call',
289
- id: this.generateId(),
263
+ id: generateId(this.name),
290
264
  model,
291
265
  timestamp,
292
266
  toolCall: {
@@ -302,11 +276,10 @@ export class GeminiAdapter extends BaseAdapter<
302
276
  }
303
277
  }
304
278
  } else if (chunk.data) {
305
- // Fallback to chunk.data if available
306
279
  accumulatedContent += chunk.data
307
280
  yield {
308
281
  type: 'content',
309
- id: this.generateId(),
282
+ id: generateId(this.name),
310
283
  model,
311
284
  timestamp,
312
285
  delta: chunk.data,
@@ -315,20 +288,14 @@ export class GeminiAdapter extends BaseAdapter<
315
288
  }
316
289
  }
317
290
 
318
- // Check for finish reason
319
291
  if (chunk.candidates?.[0]?.finishReason) {
320
292
  const finishReason = chunk.candidates[0].finishReason
321
293
 
322
- // UNEXPECTED_TOOL_CALL means Gemini tried to call a function but it wasn't properly declared
323
- // This typically means there's an issue with the tool declaration format
324
- // We should map it to tool_calls to try to process it anyway
325
294
  if (finishReason === FinishReason.UNEXPECTED_TOOL_CALL) {
326
- // Try to extract function call from content.parts if available
327
295
  if (chunk.candidates[0].content?.parts) {
328
296
  for (const part of chunk.candidates[0].content.parts) {
329
297
  const functionCall = part.functionCall
330
298
  if (functionCall) {
331
- // We found a function call - process it
332
299
  const toolCallId =
333
300
  functionCall.name || `call_${Date.now()}_${nextToolIndex}`
334
301
  const functionArgs = functionCall.args || {}
@@ -344,7 +311,7 @@ export class GeminiAdapter extends BaseAdapter<
344
311
 
345
312
  yield {
346
313
  type: 'tool_call',
347
- id: this.generateId(),
314
+ id: generateId(this.name),
348
315
  model,
349
316
  timestamp,
350
317
  toolCall: {
@@ -367,7 +334,7 @@ export class GeminiAdapter extends BaseAdapter<
367
334
  if (finishReason === FinishReason.MAX_TOKENS) {
368
335
  yield {
369
336
  type: 'error',
370
- id: this.generateId(),
337
+ id: generateId(this.name),
371
338
  model,
372
339
  timestamp,
373
340
  error: {
@@ -379,14 +346,14 @@ export class GeminiAdapter extends BaseAdapter<
379
346
 
380
347
  yield {
381
348
  type: 'done',
382
- id: this.generateId(),
349
+ id: generateId(this.name),
383
350
  model,
384
351
  timestamp,
385
352
  finishReason: toolCallMap.size > 0 ? 'tool_calls' : 'stop',
386
353
  usage: chunk.usageMetadata
387
354
  ? {
388
355
  promptTokens: chunk.usageMetadata.promptTokenCount ?? 0,
389
- completionTokens: chunk.usageMetadata.thoughtsTokenCount ?? 0,
356
+ completionTokens: chunk.usageMetadata.candidatesTokenCount ?? 0,
390
357
  totalTokens: chunk.usageMetadata.totalTokenCount ?? 0,
391
358
  }
392
359
  : undefined,
@@ -396,6 +363,20 @@ export class GeminiAdapter extends BaseAdapter<
396
363
  }
397
364
 
398
365
  private convertContentPartToGemini(part: ContentPart): Part {
366
+ const getDefaultFileType = (
367
+ part: 'image' | 'audio' | 'video' | 'document',
368
+ ) => {
369
+ switch (part) {
370
+ case 'image':
371
+ return 'image/jpeg'
372
+ case 'audio':
373
+ return 'audio/mp3'
374
+ case 'video':
375
+ return 'video/mp4'
376
+ case 'document':
377
+ return 'application/pdf'
378
+ }
379
+ }
399
380
  switch (part.type) {
400
381
  case 'text':
401
382
  return { text: part.content }
@@ -409,25 +390,23 @@ export class GeminiAdapter extends BaseAdapter<
409
390
  | GeminiVideoMetadata
410
391
  | GeminiAudioMetadata
411
392
  | undefined
412
- // Gemini uses inlineData for base64 and fileData for URLs
413
393
  if (part.source.type === 'data') {
414
394
  return {
415
395
  inlineData: {
416
396
  data: part.source.value,
417
- mimeType: metadata?.mimeType ?? 'image/jpeg',
397
+ mimeType: metadata?.mimeType ?? getDefaultFileType(part.type),
418
398
  },
419
399
  }
420
400
  } else {
421
401
  return {
422
402
  fileData: {
423
403
  fileUri: part.source.value,
424
- mimeType: metadata?.mimeType ?? 'image/jpeg',
404
+ mimeType: metadata?.mimeType ?? getDefaultFileType(part.type),
425
405
  },
426
406
  }
427
407
  }
428
408
  }
429
409
  default: {
430
- // Exhaustive check - this should never happen with known types
431
410
  const _exhaustiveCheck: never = part
432
411
  throw new Error(
433
412
  `Unsupported content part type: ${(_exhaustiveCheck as ContentPart).type}`,
@@ -443,26 +422,29 @@ export class GeminiAdapter extends BaseAdapter<
443
422
  const role: 'user' | 'model' = msg.role === 'assistant' ? 'model' : 'user'
444
423
  const parts: Array<Part> = []
445
424
 
446
- // Handle multimodal content (array of ContentPart)
447
425
  if (Array.isArray(msg.content)) {
448
426
  for (const contentPart of msg.content) {
449
427
  parts.push(this.convertContentPartToGemini(contentPart))
450
428
  }
451
429
  } else if (msg.content) {
452
- // Handle string content (backward compatibility)
453
430
  parts.push({ text: msg.content })
454
431
  }
455
432
 
456
- // Handle tool calls (from assistant)
457
433
  if (msg.role === 'assistant' && msg.toolCalls?.length) {
458
434
  for (const toolCall of msg.toolCalls) {
459
435
  let parsedArgs: Record<string, unknown> = {}
460
436
  try {
461
437
  parsedArgs = toolCall.function.arguments
462
- ? JSON.parse(toolCall.function.arguments)
438
+ ? (JSON.parse(toolCall.function.arguments) as Record<
439
+ string,
440
+ unknown
441
+ >)
463
442
  : {}
464
443
  } catch {
465
- parsedArgs = toolCall.function.arguments as any
444
+ parsedArgs = toolCall.function.arguments as unknown as Record<
445
+ string,
446
+ unknown
447
+ >
466
448
  }
467
449
 
468
450
  parts.push({
@@ -474,11 +456,10 @@ export class GeminiAdapter extends BaseAdapter<
474
456
  }
475
457
  }
476
458
 
477
- // Handle tool results (from tool role)
478
459
  if (msg.role === 'tool' && msg.toolCallId) {
479
460
  parts.push({
480
461
  functionResponse: {
481
- name: msg.toolCallId, // Gemini uses function name here
462
+ name: msg.toolCallId,
482
463
  response: {
483
464
  content: msg.content || '',
484
465
  },
@@ -492,22 +473,20 @@ export class GeminiAdapter extends BaseAdapter<
492
473
  }
493
474
  })
494
475
  }
495
- /**
496
- * Maps common options to Gemini-specific format
497
- * Handles translation of normalized options to Gemini's API format
498
- */
499
- private mapCommonOptionsToGemini(options: ChatOptions) {
500
- const providerOpts = options.providerOptions
476
+
477
+ private mapCommonOptionsToGemini(options: TextOptions) {
478
+ const providerOpts = options.modelOptions
501
479
  const requestOptions: GenerateContentParameters = {
502
480
  model: options.model,
503
481
  contents: this.formatMessages(options.messages),
504
482
  config: {
505
483
  ...providerOpts,
506
- temperature: options.options?.temperature,
507
- topP: options.options?.topP,
508
- maxOutputTokens: options.options?.maxTokens,
484
+ temperature: options.temperature,
485
+ topP: options.topP,
486
+ maxOutputTokens: options.maxTokens,
509
487
  systemInstruction: options.systemPrompts?.join('\n'),
510
- ...providerOpts?.generationConfig,
488
+ ...((providerOpts as Record<string, unknown> | undefined)
489
+ ?.generationConfig as Record<string, unknown> | undefined),
511
490
  tools: convertToolsToProviderFormat(options.tools),
512
491
  },
513
492
  }
@@ -517,61 +496,33 @@ export class GeminiAdapter extends BaseAdapter<
517
496
  }
518
497
 
519
498
  /**
520
- * Creates a Gemini adapter with simplified configuration
521
- * @param apiKey - Your Google API key
522
- * @returns A fully configured Gemini adapter instance
523
- *
524
- * @example
525
- * ```typescript
526
- * const gemini = createGemini("AIza...");
527
- *
528
- * const ai = new AI({
529
- * adapters: {
530
- * gemini,
531
- * }
532
- * });
533
- * ```
499
+ * Creates a Gemini text adapter with explicit API key.
500
+ * Type resolution happens here at the call site.
534
501
  */
535
- export function createGemini(
502
+ export function createGeminiChat<TModel extends (typeof GEMINI_MODELS)[number]>(
503
+ model: TModel,
536
504
  apiKey: string,
537
- config?: Omit<GeminiAdapterConfig, 'apiKey'>,
538
- ): GeminiAdapter {
539
- return new GeminiAdapter({ apiKey, ...config })
505
+ config?: Omit<GeminiTextConfig, 'apiKey'>,
506
+ ): GeminiTextAdapter<
507
+ TModel,
508
+ ResolveProviderOptions<TModel>,
509
+ ResolveInputModalities<TModel>
510
+ > {
511
+ return new GeminiTextAdapter({ apiKey, ...config }, model)
540
512
  }
541
513
 
542
514
  /**
543
- * Create a Gemini adapter with automatic API key detection from environment variables.
544
- *
545
- * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:
546
- * - `process.env` (Node.js)
547
- * - `window.env` (Browser with injected env)
548
- *
549
- * @param config - Optional configuration (excluding apiKey which is auto-detected)
550
- * @returns Configured Gemini adapter instance
551
- * @throws Error if API key is not found in environment
552
- *
553
- * @example
554
- * ```typescript
555
- * // Automatically uses GOOGLE_API_KEY or GEMINI_API_KEY from environment
556
- * const aiInstance = ai(gemini());
557
- * ```
515
+ * Creates a Gemini text adapter with automatic API key detection.
516
+ * Type resolution happens here at the call site.
558
517
  */
559
- export function gemini(
560
- config?: Omit<GeminiAdapterConfig, 'apiKey'>,
561
- ): GeminiAdapter {
562
- const env =
563
- typeof globalThis !== 'undefined' && (globalThis as any).window?.env
564
- ? (globalThis as any).window.env
565
- : typeof process !== 'undefined'
566
- ? process.env
567
- : undefined
568
- const key = env?.GOOGLE_API_KEY || env?.GEMINI_API_KEY
569
-
570
- if (!key) {
571
- throw new Error(
572
- 'GOOGLE_API_KEY or GEMINI_API_KEY is required. Please set it in your environment variables or use createGemini(apiKey, config) instead.',
573
- )
574
- }
575
-
576
- return createGemini(key, config)
518
+ export function geminiText<TModel extends (typeof GEMINI_MODELS)[number]>(
519
+ model: TModel,
520
+ config?: Omit<GeminiTextConfig, 'apiKey'>,
521
+ ): GeminiTextAdapter<
522
+ TModel,
523
+ ResolveProviderOptions<TModel>,
524
+ ResolveInputModalities<TModel>
525
+ > {
526
+ const apiKey = getGeminiApiKeyFromEnv()
527
+ return createGeminiChat(model, apiKey, config)
577
528
  }