@tanstack/ai-gemini 0.0.3 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +26 -0
  2. package/dist/esm/adapters/image.d.ts +83 -0
  3. package/dist/esm/adapters/image.js +60 -0
  4. package/dist/esm/adapters/image.js.map +1 -0
  5. package/dist/esm/adapters/summarize.d.ts +53 -0
  6. package/dist/esm/adapters/summarize.js +139 -0
  7. package/dist/esm/adapters/summarize.js.map +1 -0
  8. package/dist/esm/adapters/text.d.ts +63 -0
  9. package/dist/esm/{gemini-adapter.js → adapters/text.js} +86 -132
  10. package/dist/esm/adapters/text.js.map +1 -0
  11. package/dist/esm/adapters/tts.d.ts +129 -0
  12. package/dist/esm/adapters/tts.js +78 -0
  13. package/dist/esm/adapters/tts.js.map +1 -0
  14. package/dist/esm/image/image-provider-options.d.ts +139 -0
  15. package/dist/esm/image/image-provider-options.js +53 -0
  16. package/dist/esm/image/image-provider-options.js.map +1 -0
  17. package/dist/esm/index.d.ts +15 -2
  18. package/dist/esm/index.js +22 -4
  19. package/dist/esm/index.js.map +1 -1
  20. package/dist/esm/model-meta.d.ts +128 -1
  21. package/dist/esm/model-meta.js +71 -5
  22. package/dist/esm/model-meta.js.map +1 -1
  23. package/dist/esm/tools/tool-converter.js +5 -3
  24. package/dist/esm/tools/tool-converter.js.map +1 -1
  25. package/dist/esm/utils/client.d.ts +17 -0
  26. package/dist/esm/utils/client.js +25 -0
  27. package/dist/esm/utils/client.js.map +1 -0
  28. package/dist/esm/utils/index.d.ts +1 -0
  29. package/package.json +5 -4
  30. package/src/adapters/image.ts +188 -0
  31. package/src/adapters/summarize.ts +242 -0
  32. package/src/{gemini-adapter.ts → adapters/text.ts} +196 -245
  33. package/src/adapters/tts.ts +223 -0
  34. package/src/image/image-provider-options.ts +239 -0
  35. package/src/index.ts +66 -2
  36. package/src/model-meta.ts +87 -62
  37. package/src/tools/tool-converter.ts +6 -5
  38. package/src/utils/client.ts +43 -0
  39. package/src/utils/index.ts +6 -0
  40. package/dist/esm/gemini-adapter.d.ts +0 -71
  41. package/dist/esm/gemini-adapter.js.map +0 -1
@@ -0,0 +1,223 @@
1
+ import { BaseTTSAdapter } from '@tanstack/ai/adapters'
2
+ import {
3
+ createGeminiClient,
4
+ generateId,
5
+ getGeminiApiKeyFromEnv,
6
+ } from '../utils'
7
+ import type { GEMINI_TTS_MODELS, GeminiTTSVoice } from '../model-meta'
8
+ import type { TTSOptions, TTSResult } from '@tanstack/ai'
9
+ import type { GoogleGenAI } from '@google/genai'
10
+ import type { GeminiClientConfig } from '../utils'
11
+
12
+ /**
13
+ * Provider-specific options for Gemini TTS
14
+ *
15
+ * @experimental Gemini TTS is an experimental feature.
16
+ * @see https://ai.google.dev/gemini-api/docs/speech-generation
17
+ */
18
+ export interface GeminiTTSProviderOptions {
19
+ /**
20
+ * Voice configuration for TTS.
21
+ * Choose from 30 available voices with different characteristics.
22
+ */
23
+ voiceConfig?: {
24
+ prebuiltVoiceConfig?: {
25
+ /**
26
+ * The voice name to use for speech synthesis.
27
+ * @see https://ai.google.dev/gemini-api/docs/speech-generation#voices
28
+ */
29
+ voiceName?: GeminiTTSVoice
30
+ }
31
+ }
32
+
33
+ /**
34
+ * System instruction for controlling speech style.
35
+ * Use natural language to describe the desired speaking style,
36
+ * pace, tone, accent, or other characteristics.
37
+ *
38
+ * @example "Speak slowly and calmly, as if telling a bedtime story"
39
+ * @example "Use an upbeat, enthusiastic tone with moderate pace"
40
+ * @example "Speak with a British accent"
41
+ */
42
+ systemInstruction?: string
43
+
44
+ /**
45
+ * Language code hint for the speech synthesis.
46
+ * Gemini TTS supports 24 languages and can auto-detect,
47
+ * but you can provide a hint for better results.
48
+ *
49
+ * @example "en-US" for American English
50
+ * @example "es-ES" for Spanish (Spain)
51
+ * @example "ja-JP" for Japanese
52
+ */
53
+ languageCode?: string
54
+ }
55
+
56
+ /**
57
+ * Configuration for Gemini TTS adapter
58
+ *
59
+ * @experimental Gemini TTS is an experimental feature.
60
+ */
61
+ export interface GeminiTTSConfig extends GeminiClientConfig {}
62
+
63
+ /** Model type for Gemini TTS */
64
+ export type GeminiTTSModel = (typeof GEMINI_TTS_MODELS)[number]
65
+
66
+ /**
67
+ * Gemini Text-to-Speech Adapter
68
+ *
69
+ * Tree-shakeable adapter for Gemini TTS functionality.
70
+ *
71
+ * **IMPORTANT**: Gemini TTS uses the Live API (WebSocket-based) which requires
72
+ * different handling than traditional REST APIs. This adapter provides a
73
+ * simplified interface but may have limitations.
74
+ *
75
+ * @experimental Gemini TTS is an experimental feature and may change.
76
+ *
77
+ * Models:
78
+ * - gemini-2.5-flash-preview-tts
79
+ */
80
+ export class GeminiTTSAdapter<
81
+ TModel extends GeminiTTSModel,
82
+ > extends BaseTTSAdapter<TModel, GeminiTTSProviderOptions> {
83
+ readonly name = 'gemini' as const
84
+
85
+ private client: GoogleGenAI
86
+
87
+ constructor(config: GeminiTTSConfig, model: TModel) {
88
+ super(config, model)
89
+ this.client = createGeminiClient(config)
90
+ }
91
+
92
+ /**
93
+ * Generate speech from text using Gemini's TTS model.
94
+ *
95
+ * @experimental This implementation is experimental and may change.
96
+ * @see https://ai.google.dev/gemini-api/docs/speech-generation
97
+ */
98
+ async generateSpeech(
99
+ options: TTSOptions<GeminiTTSProviderOptions>,
100
+ ): Promise<TTSResult> {
101
+ const { model, text, modelOptions } = options
102
+
103
+ const voiceConfig = modelOptions?.voiceConfig || {
104
+ prebuiltVoiceConfig: {
105
+ voiceName: 'Kore',
106
+ },
107
+ }
108
+
109
+ const response = await this.client.models.generateContent({
110
+ model,
111
+ contents: [
112
+ {
113
+ role: 'user',
114
+ parts: [{ text }],
115
+ },
116
+ ],
117
+ config: {
118
+ responseModalities: ['AUDIO'],
119
+ speechConfig: {
120
+ voiceConfig,
121
+ ...(modelOptions?.languageCode && {
122
+ languageCode: modelOptions.languageCode,
123
+ }),
124
+ },
125
+ },
126
+ ...(modelOptions?.systemInstruction && {
127
+ systemInstruction: modelOptions.systemInstruction,
128
+ }),
129
+ })
130
+
131
+ // Extract audio data from response
132
+ const candidate = response.candidates?.[0]
133
+ const parts = candidate?.content?.parts
134
+
135
+ if (!parts || parts.length === 0) {
136
+ throw new Error('No audio output received from Gemini TTS')
137
+ }
138
+
139
+ // Look for inline data (audio)
140
+ const audioPart = parts.find((part: any) =>
141
+ part.inlineData?.mimeType?.startsWith('audio/'),
142
+ )
143
+
144
+ if (!audioPart || !audioPart.inlineData || !audioPart.inlineData.data) {
145
+ throw new Error('No audio data in Gemini TTS response')
146
+ }
147
+
148
+ const audioBase64 = audioPart.inlineData.data
149
+ const mimeType = audioPart.inlineData.mimeType || 'audio/wav'
150
+ const format = mimeType.split('/')[1] || 'wav'
151
+
152
+ return {
153
+ id: generateId(this.name),
154
+ model,
155
+ audio: audioBase64,
156
+ format,
157
+ contentType: mimeType,
158
+ }
159
+ }
160
+ }
161
+
162
+ /**
163
+ * Creates a Gemini TTS adapter with explicit API key.
164
+ * Type resolution happens here at the call site.
165
+ *
166
+ * @experimental Gemini TTS is an experimental feature and may change.
167
+ *
168
+ * @param model - The model name (e.g., 'gemini-2.5-flash-preview-tts')
169
+ * @param apiKey - Your Google API key
170
+ * @param config - Optional additional configuration
171
+ * @returns Configured Gemini TTS adapter instance with resolved types
172
+ *
173
+ * @example
174
+ * ```typescript
175
+ * const adapter = createGeminiSpeech('gemini-2.5-flash-preview-tts', "your-api-key");
176
+ *
177
+ * const result = await generateSpeech({
178
+ * adapter,
179
+ * text: 'Hello, world!'
180
+ * });
181
+ * ```
182
+ */
183
+ export function createGeminiSpeech<TModel extends GeminiTTSModel>(
184
+ model: TModel,
185
+ apiKey: string,
186
+ config?: Omit<GeminiTTSConfig, 'apiKey'>,
187
+ ): GeminiTTSAdapter<TModel> {
188
+ return new GeminiTTSAdapter({ apiKey, ...config }, model)
189
+ }
190
+
191
+ /**
192
+ * Creates a Gemini speech adapter with automatic API key detection from environment variables.
193
+ * Type resolution happens here at the call site.
194
+ *
195
+ * @experimental Gemini TTS is an experimental feature and may change.
196
+ *
197
+ * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:
198
+ * - `process.env` (Node.js)
199
+ * - `window.env` (Browser with injected env)
200
+ *
201
+ * @param model - The model name (e.g., 'gemini-2.5-flash-preview-tts')
202
+ * @param config - Optional configuration (excluding apiKey which is auto-detected)
203
+ * @returns Configured Gemini speech adapter instance with resolved types
204
+ * @throws Error if GOOGLE_API_KEY or GEMINI_API_KEY is not found in environment
205
+ *
206
+ * @example
207
+ * ```typescript
208
+ * // Automatically uses GOOGLE_API_KEY from environment
209
+ * const adapter = geminiSpeech('gemini-2.5-flash-preview-tts');
210
+ *
211
+ * const result = await generateSpeech({
212
+ * adapter,
213
+ * text: 'Welcome to TanStack AI!'
214
+ * });
215
+ * ```
216
+ */
217
+ export function geminiSpeech<TModel extends GeminiTTSModel>(
218
+ model: TModel,
219
+ config?: Omit<GeminiTTSConfig, 'apiKey'>,
220
+ ): GeminiTTSAdapter<TModel> {
221
+ const apiKey = getGeminiApiKeyFromEnv()
222
+ return createGeminiSpeech(model, apiKey, config)
223
+ }
@@ -0,0 +1,239 @@
1
+ import type { GeminiImageModels } from '../model-meta'
2
+ import type {
3
+ ImagePromptLanguage,
4
+ PersonGeneration,
5
+ SafetyFilterLevel,
6
+ } from '@google/genai'
7
+
8
+ // Re-export SDK types so users can use them directly
9
+ export type { ImagePromptLanguage, PersonGeneration, SafetyFilterLevel }
10
+
11
+ /**
12
+ * Gemini Imagen aspect ratio options
13
+ * Controls the aspect ratio of generated images
14
+ */
15
+ export type GeminiAspectRatio =
16
+ | '1:1'
17
+ | '3:4'
18
+ | '4:3'
19
+ | '9:16'
20
+ | '16:9'
21
+ | '9:21'
22
+ | '21:9'
23
+
24
+ /**
25
+ * Provider options for Gemini image generation
26
+ * These options match the @google/genai GenerateImagesConfig interface
27
+ * and can be spread directly into the API request.
28
+ */
29
+ export interface GeminiImageProviderOptions {
30
+ /**
31
+ * The aspect ratio of generated images
32
+ * @default '1:1'
33
+ */
34
+ aspectRatio?: GeminiAspectRatio
35
+
36
+ /**
37
+ * Controls whether people can appear in generated images
38
+ * Use PersonGeneration enum values: DONT_ALLOW, ALLOW_ADULT, ALLOW_ALL
39
+ * @default 'ALLOW_ADULT'
40
+ */
41
+ personGeneration?: PersonGeneration
42
+
43
+ /**
44
+ * Safety filter level for content filtering
45
+ * Use SafetyFilterLevel enum values
46
+ */
47
+ safetyFilterLevel?: SafetyFilterLevel
48
+
49
+ /**
50
+ * Optional seed for reproducible image generation
51
+ * When the same seed is used with the same prompt and settings,
52
+ * you should get similar (though not identical) results
53
+ */
54
+ seed?: number
55
+
56
+ /**
57
+ * Whether to add a SynthID watermark to generated images
58
+ * SynthID helps identify AI-generated content
59
+ * @default true
60
+ */
61
+ addWatermark?: boolean
62
+
63
+ /**
64
+ * Language of the prompt
65
+ * Use ImagePromptLanguage enum values
66
+ */
67
+ language?: ImagePromptLanguage
68
+
69
+ /**
70
+ * Negative prompt - what to avoid in the generated image
71
+ * Not all models support negative prompts
72
+ */
73
+ negativePrompt?: string
74
+
75
+ /**
76
+ * Output MIME type for the generated image
77
+ * @default 'image/png'
78
+ */
79
+ outputMimeType?: 'image/png' | 'image/jpeg' | 'image/webp'
80
+
81
+ /**
82
+ * Compression quality for JPEG outputs (0-100)
83
+ * Higher values mean better quality but larger file sizes
84
+ * @default 75
85
+ */
86
+ outputCompressionQuality?: number
87
+
88
+ /**
89
+ * Controls how much the model adheres to the text prompt
90
+ * Large values increase output and prompt alignment,
91
+ * but may compromise image quality
92
+ */
93
+ guidanceScale?: number
94
+
95
+ /**
96
+ * Whether to use the prompt rewriting logic
97
+ */
98
+ enhancePrompt?: boolean
99
+
100
+ /**
101
+ * Whether to report the safety scores of each generated image
102
+ * and the positive prompt in the response
103
+ */
104
+ includeSafetyAttributes?: boolean
105
+
106
+ /**
107
+ * Whether to include the Responsible AI filter reason
108
+ * if the image is filtered out of the response
109
+ */
110
+ includeRaiReason?: boolean
111
+
112
+ /**
113
+ * Cloud Storage URI used to store the generated images
114
+ */
115
+ outputGcsUri?: string
116
+
117
+ /**
118
+ * User specified labels to track billing usage
119
+ */
120
+ labels?: Record<string, string>
121
+ }
122
+
123
+ /**
124
+ * Model-specific provider options mapping
125
+ * Currently all Imagen models use the same options structure
126
+ */
127
+ export type GeminiImageModelProviderOptionsByName = {
128
+ [K in GeminiImageModels]: GeminiImageProviderOptions
129
+ }
130
+
131
+ /**
132
+ * Supported size strings for Gemini Imagen models
133
+ * These map to aspect ratios internally
134
+ */
135
+ export type GeminiImageSize =
136
+ | '1024x1024'
137
+ | '512x512'
138
+ | '1024x768'
139
+ | '1536x1024'
140
+ | '1792x1024'
141
+ | '1920x1080'
142
+ | '768x1024'
143
+ | '1024x1536'
144
+ | '1024x1792'
145
+ | '1080x1920'
146
+
147
+ /**
148
+ * Model-specific size options mapping
149
+ * All Imagen models use the same size options
150
+ */
151
+ export type GeminiImageModelSizeByName = {
152
+ [K in GeminiImageModels]: GeminiImageSize
153
+ }
154
+
155
+ /**
156
+ * Valid sizes for Gemini Imagen models
157
+ * Gemini uses aspect ratios, but we map common WIDTHxHEIGHT formats to aspect ratios
158
+ * These are approximate mappings based on common image dimensions
159
+ */
160
+ export const GEMINI_SIZE_TO_ASPECT_RATIO: Record<string, GeminiAspectRatio> = {
161
+ // Square
162
+ '1024x1024': '1:1',
163
+ '512x512': '1:1',
164
+ // Landscape
165
+ '1024x768': '4:3',
166
+ '1536x1024': '4:3',
167
+ '1792x1024': '16:9',
168
+ '1920x1080': '16:9',
169
+ // Portrait
170
+ '768x1024': '3:4',
171
+ '1024x1536': '3:4', // Inverted
172
+ '1024x1792': '9:16',
173
+ '1080x1920': '9:16',
174
+ }
175
+
176
+ /**
177
+ * Maps a WIDTHxHEIGHT size string to a Gemini aspect ratio
178
+ * Returns undefined if the size cannot be mapped
179
+ */
180
+ export function sizeToAspectRatio(
181
+ size: string | undefined,
182
+ ): GeminiAspectRatio | undefined {
183
+ if (!size) return undefined
184
+ return GEMINI_SIZE_TO_ASPECT_RATIO[size]
185
+ }
186
+
187
+ /**
188
+ * Validates that the provided size can be mapped to an aspect ratio
189
+ * Throws an error if the size is invalid
190
+ */
191
+ export function validateImageSize(
192
+ model: string,
193
+ size: string | undefined,
194
+ ): void {
195
+ if (!size) return
196
+
197
+ const aspectRatio = sizeToAspectRatio(size)
198
+ if (!aspectRatio) {
199
+ const validSizes = Object.keys(GEMINI_SIZE_TO_ASPECT_RATIO)
200
+ throw new Error(
201
+ `Invalid size "${size}" for model "${model}". ` +
202
+ `Gemini Imagen uses aspect ratios. Valid sizes that map to aspect ratios: ${validSizes.join(', ')}. ` +
203
+ `Alternatively, use providerOptions.aspectRatio directly with values: 1:1, 3:4, 4:3, 9:16, 16:9, 9:21, 21:9`,
204
+ )
205
+ }
206
+ }
207
+
208
+ /**
209
+ * Validates the number of images requested
210
+ * Imagen models support 1-8 images per request (varies by model)
211
+ */
212
+ export function validateNumberOfImages(
213
+ model: string,
214
+ numberOfImages: number | undefined,
215
+ ): void {
216
+ if (numberOfImages === undefined) return
217
+
218
+ // Most Imagen models support 1-4 images, some support up to 8
219
+ const maxImages = 4
220
+ if (numberOfImages < 1 || numberOfImages > maxImages) {
221
+ throw new Error(
222
+ `Invalid numberOfImages "${numberOfImages}" for model "${model}". ` +
223
+ `Must be between 1 and ${maxImages}.`,
224
+ )
225
+ }
226
+ }
227
+
228
+ /**
229
+ * Validates the prompt is not empty
230
+ */
231
+ export function validatePrompt(options: {
232
+ prompt: string
233
+ model: string
234
+ }): void {
235
+ const { prompt, model } = options
236
+ if (!prompt || prompt.trim().length === 0) {
237
+ throw new Error(`Prompt cannot be empty for model "${model}".`)
238
+ }
239
+ }
package/src/index.ts CHANGED
@@ -1,5 +1,69 @@
1
- export { GeminiAdapter, createGemini, gemini } from './gemini-adapter'
2
- export type { GeminiAdapterConfig } from './gemini-adapter'
1
+ // ===========================
2
+ // New tree-shakeable adapters
3
+ // ===========================
4
+
5
+ // Text/Chat adapter
6
+ export {
7
+ GeminiTextAdapter,
8
+ createGeminiChat,
9
+ geminiText,
10
+ type GeminiTextConfig,
11
+ type GeminiTextProviderOptions,
12
+ } from './adapters/text'
13
+
14
+ // Summarize adapter
15
+ export {
16
+ GeminiSummarizeAdapter,
17
+ GeminiSummarizeModels,
18
+ createGeminiSummarize,
19
+ geminiSummarize,
20
+ type GeminiSummarizeAdapterOptions,
21
+ type GeminiSummarizeModel,
22
+ type GeminiSummarizeProviderOptions,
23
+ } from './adapters/summarize'
24
+
25
+ // Image adapter
26
+ export {
27
+ GeminiImageAdapter,
28
+ createGeminiImage,
29
+ geminiImage,
30
+ type GeminiImageConfig,
31
+ } from './adapters/image'
32
+ export type {
33
+ GeminiImageProviderOptions,
34
+ GeminiImageModelProviderOptionsByName,
35
+ GeminiAspectRatio,
36
+ // Re-export SDK types for convenience
37
+ PersonGeneration,
38
+ SafetyFilterLevel,
39
+ ImagePromptLanguage,
40
+ } from './image/image-provider-options'
41
+
42
+ // TTS adapter (experimental)
43
+ /**
44
+ * @experimental Gemini TTS is an experimental feature and may change.
45
+ */
46
+ export {
47
+ GeminiTTSAdapter,
48
+ createGeminiSpeech,
49
+ geminiSpeech,
50
+ type GeminiTTSConfig,
51
+ type GeminiTTSProviderOptions,
52
+ } from './adapters/tts'
53
+
54
+ // Re-export models from model-meta for convenience
55
+ export { GEMINI_MODELS as GeminiTextModels } from './model-meta'
56
+ export { GEMINI_IMAGE_MODELS as GeminiImageModels } from './model-meta'
57
+ export { GEMINI_TTS_MODELS as GeminiTTSModels } from './model-meta'
58
+ export { GEMINI_TTS_VOICES as GeminiTTSVoices } from './model-meta'
59
+ export type { GeminiModels as GeminiTextModel } from './model-meta'
60
+ export type { GeminiImageModels as GeminiImageModel } from './model-meta'
61
+ export type { GeminiTTSVoice } from './model-meta'
62
+
63
+ // ===========================
64
+ // Type Exports
65
+ // ===========================
66
+
3
67
  export type {
4
68
  GeminiChatModelProviderOptionsByName,
5
69
  GeminiModelInputModalitiesByName,