@tanstack/ai-gemini 0.17.2 → 0.18.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/package.json +5 -4
  2. package/src/experimental/text-interactions/adapter.ts +304 -100
  3. package/src/experimental/text-interactions/events.ts +29 -23
  4. package/src/experimental/text-interactions/provider-options.ts +0 -1
  5. package/dist/esm/adapters/audio.d.ts +0 -92
  6. package/dist/esm/adapters/audio.js +0 -66
  7. package/dist/esm/adapters/audio.js.map +0 -1
  8. package/dist/esm/adapters/image.d.ts +0 -101
  9. package/dist/esm/adapters/image.js +0 -253
  10. package/dist/esm/adapters/image.js.map +0 -1
  11. package/dist/esm/adapters/summarize.d.ts +0 -33
  12. package/dist/esm/adapters/summarize.js +0 -18
  13. package/dist/esm/adapters/summarize.js.map +0 -1
  14. package/dist/esm/adapters/text.d.ts +0 -85
  15. package/dist/esm/adapters/text.js +0 -658
  16. package/dist/esm/adapters/text.js.map +0 -1
  17. package/dist/esm/adapters/tts.d.ts +0 -165
  18. package/dist/esm/adapters/tts.js +0 -192
  19. package/dist/esm/adapters/tts.js.map +0 -1
  20. package/dist/esm/adapters/video.d.ts +0 -108
  21. package/dist/esm/adapters/video.js +0 -227
  22. package/dist/esm/adapters/video.js.map +0 -1
  23. package/dist/esm/experimental/index.d.ts +0 -6
  24. package/dist/esm/experimental/index.js +0 -7
  25. package/dist/esm/experimental/index.js.map +0 -1
  26. package/dist/esm/experimental/text-interactions/adapter.d.ts +0 -75
  27. package/dist/esm/experimental/text-interactions/adapter.js +0 -899
  28. package/dist/esm/experimental/text-interactions/adapter.js.map +0 -1
  29. package/dist/esm/experimental/text-interactions/events.d.ts +0 -141
  30. package/dist/esm/experimental/text-interactions/provider-options.d.ts +0 -15
  31. package/dist/esm/image/image-provider-options.d.ts +0 -181
  32. package/dist/esm/image/image-provider-options.js +0 -67
  33. package/dist/esm/image/image-provider-options.js.map +0 -1
  34. package/dist/esm/index.d.ts +0 -33
  35. package/dist/esm/index.js +0 -38
  36. package/dist/esm/index.js.map +0 -1
  37. package/dist/esm/message-types.d.ts +0 -104
  38. package/dist/esm/model-meta.d.ts +0 -242
  39. package/dist/esm/model-meta.js +0 -159
  40. package/dist/esm/model-meta.js.map +0 -1
  41. package/dist/esm/text/text-provider-options.d.ts +0 -196
  42. package/dist/esm/tools/code-execution-tool.d.ts +0 -10
  43. package/dist/esm/tools/code-execution-tool.js +0 -18
  44. package/dist/esm/tools/code-execution-tool.js.map +0 -1
  45. package/dist/esm/tools/computer-use-tool.d.ts +0 -13
  46. package/dist/esm/tools/computer-use-tool.js +0 -33
  47. package/dist/esm/tools/computer-use-tool.js.map +0 -1
  48. package/dist/esm/tools/file-search-tool.d.ts +0 -10
  49. package/dist/esm/tools/file-search-tool.js +0 -19
  50. package/dist/esm/tools/file-search-tool.js.map +0 -1
  51. package/dist/esm/tools/function-declaration-tool.d.ts +0 -5
  52. package/dist/esm/tools/function-declaration-tool.js +0 -28
  53. package/dist/esm/tools/function-declaration-tool.js.map +0 -1
  54. package/dist/esm/tools/google-maps-tool.d.ts +0 -10
  55. package/dist/esm/tools/google-maps-tool.js +0 -19
  56. package/dist/esm/tools/google-maps-tool.js.map +0 -1
  57. package/dist/esm/tools/google-search-retriveal-tool.d.ts +0 -10
  58. package/dist/esm/tools/google-search-retriveal-tool.js +0 -19
  59. package/dist/esm/tools/google-search-retriveal-tool.js.map +0 -1
  60. package/dist/esm/tools/google-search-tool.d.ts +0 -10
  61. package/dist/esm/tools/google-search-tool.js +0 -19
  62. package/dist/esm/tools/google-search-tool.js.map +0 -1
  63. package/dist/esm/tools/index.d.ts +0 -18
  64. package/dist/esm/tools/index.js +0 -21
  65. package/dist/esm/tools/index.js.map +0 -1
  66. package/dist/esm/tools/tool-converter.d.ts +0 -22
  67. package/dist/esm/tools/tool-converter.js +0 -66
  68. package/dist/esm/tools/tool-converter.js.map +0 -1
  69. package/dist/esm/tools/url-context-tool.d.ts +0 -10
  70. package/dist/esm/tools/url-context-tool.js +0 -18
  71. package/dist/esm/tools/url-context-tool.js.map +0 -1
  72. package/dist/esm/usage.d.ts +0 -67
  73. package/dist/esm/usage.js +0 -94
  74. package/dist/esm/usage.js.map +0 -1
  75. package/dist/esm/utils/client.d.ts +0 -17
  76. package/dist/esm/utils/client.js +0 -30
  77. package/dist/esm/utils/client.js.map +0 -1
  78. package/dist/esm/utils/index.d.ts +0 -1
  79. package/dist/esm/video/video-provider-options.d.ts +0 -90
  80. package/dist/esm/video/video-provider-options.js +0 -15
  81. package/dist/esm/video/video-provider-options.js.map +0 -1
@@ -15,55 +15,60 @@ export interface GeminiInteractionIdEvent extends Omit<
15
15
  }
16
16
 
17
17
  /**
18
- * `CUSTOM` event carrying a raw `google_search_call` content delta from the
18
+ * `CUSTOM` event carrying a raw `google_search_call` Step from the
19
19
  * Interactions API. Payload shape is owned by `@google/genai`.
20
+ *
21
+ * SDK 2.x: server-side tool activity is now first-class `Step`s (emitted
22
+ * via `step.start`) rather than content deltas, so each variant below
23
+ * carries the full Step (with its `id` / `call_id`) instead of a delta
24
+ * fragment.
20
25
  */
21
26
  export interface GeminiGoogleSearchCallEvent extends Omit<
22
27
  CustomEvent,
23
28
  'name' | 'value'
24
29
  > {
25
30
  name: 'gemini.googleSearchCall'
26
- value: Interactions.ContentDelta.GoogleSearchCallDelta
31
+ value: Interactions.GoogleSearchCallStep
27
32
  }
28
33
 
29
34
  /**
30
- * `CUSTOM` event carrying a raw `google_search_result` content delta from
31
- * the Interactions API.
35
+ * `CUSTOM` event carrying a raw `google_search_result` Step from the
36
+ * Interactions API.
32
37
  */
33
38
  export interface GeminiGoogleSearchResultEvent extends Omit<
34
39
  CustomEvent,
35
40
  'name' | 'value'
36
41
  > {
37
42
  name: 'gemini.googleSearchResult'
38
- value: Interactions.ContentDelta.GoogleSearchResultDelta
43
+ value: Interactions.GoogleSearchResultStep
39
44
  }
40
45
 
41
46
  /**
42
- * `CUSTOM` event carrying a raw `code_execution_call` content delta from
43
- * the Interactions API.
47
+ * `CUSTOM` event carrying a raw `code_execution_call` Step from the
48
+ * Interactions API.
44
49
  */
45
50
  export interface GeminiCodeExecutionCallEvent extends Omit<
46
51
  CustomEvent,
47
52
  'name' | 'value'
48
53
  > {
49
54
  name: 'gemini.codeExecutionCall'
50
- value: Interactions.ContentDelta.CodeExecutionCallDelta
55
+ value: Interactions.CodeExecutionCallStep
51
56
  }
52
57
 
53
58
  /**
54
- * `CUSTOM` event carrying a raw `code_execution_result` content delta from
55
- * the Interactions API.
59
+ * `CUSTOM` event carrying a raw `code_execution_result` Step from the
60
+ * Interactions API.
56
61
  */
57
62
  export interface GeminiCodeExecutionResultEvent extends Omit<
58
63
  CustomEvent,
59
64
  'name' | 'value'
60
65
  > {
61
66
  name: 'gemini.codeExecutionResult'
62
- value: Interactions.ContentDelta.CodeExecutionResultDelta
67
+ value: Interactions.CodeExecutionResultStep
63
68
  }
64
69
 
65
70
  /**
66
- * `CUSTOM` event carrying a raw `url_context_call` content delta from the
71
+ * `CUSTOM` event carrying a raw `url_context_call` Step from the
67
72
  * Interactions API.
68
73
  */
69
74
  export interface GeminiUrlContextCallEvent extends Omit<
@@ -71,11 +76,11 @@ export interface GeminiUrlContextCallEvent extends Omit<
71
76
  'name' | 'value'
72
77
  > {
73
78
  name: 'gemini.urlContextCall'
74
- value: Interactions.ContentDelta.URLContextCallDelta
79
+ value: Interactions.URLContextCallStep
75
80
  }
76
81
 
77
82
  /**
78
- * `CUSTOM` event carrying a raw `url_context_result` content delta from the
83
+ * `CUSTOM` event carrying a raw `url_context_result` Step from the
79
84
  * Interactions API.
80
85
  */
81
86
  export interface GeminiUrlContextResultEvent extends Omit<
@@ -83,11 +88,11 @@ export interface GeminiUrlContextResultEvent extends Omit<
83
88
  'name' | 'value'
84
89
  > {
85
90
  name: 'gemini.urlContextResult'
86
- value: Interactions.ContentDelta.URLContextResultDelta
91
+ value: Interactions.URLContextResultStep
87
92
  }
88
93
 
89
94
  /**
90
- * `CUSTOM` event carrying a raw `file_search_call` content delta from the
95
+ * `CUSTOM` event carrying a raw `file_search_call` Step from the
91
96
  * Interactions API.
92
97
  */
93
98
  export interface GeminiFileSearchCallEvent extends Omit<
@@ -95,11 +100,11 @@ export interface GeminiFileSearchCallEvent extends Omit<
95
100
  'name' | 'value'
96
101
  > {
97
102
  name: 'gemini.fileSearchCall'
98
- value: Interactions.ContentDelta.FileSearchCallDelta
103
+ value: Interactions.FileSearchCallStep
99
104
  }
100
105
 
101
106
  /**
102
- * `CUSTOM` event carrying a raw `file_search_result` content delta from the
107
+ * `CUSTOM` event carrying a raw `file_search_result` Step from the
103
108
  * Interactions API.
104
109
  */
105
110
  export interface GeminiFileSearchResultEvent extends Omit<
@@ -107,7 +112,7 @@ export interface GeminiFileSearchResultEvent extends Omit<
107
112
  'name' | 'value'
108
113
  > {
109
114
  name: 'gemini.fileSearchResult'
110
- value: Interactions.ContentDelta.FileSearchResultDelta
115
+ value: Interactions.FileSearchResultStep
111
116
  }
112
117
 
113
118
  /**
@@ -127,12 +132,13 @@ export interface GeminiFileSearchResultEvent extends Omit<
127
132
  * }
128
133
  * ```
129
134
  *
130
- * The four tool-delta variant pairs (`google_search`, `code_execution`,
135
+ * The four tool variant pairs (`google_search`, `code_execution`,
131
136
  * `url_context`, `file_search`) forward the raw
132
- * `Interactions.ContentDelta.*Delta` payload from `@google/genai` so
133
- * consumers stay in sync with SDK shape changes automatically.
137
+ * `Interactions.*CallStep` / `Interactions.*ResultStep` payload from
138
+ * `@google/genai`. SDK 2.x emits these as full Steps via `step.start`
139
+ * (carrying the call's `id` / `call_id`) rather than as content deltas.
134
140
  * `computer_use` is accepted as a request tool but the API does not
135
- * stream per-delta CUSTOM events for it.
141
+ * surface a dedicated CUSTOM event for it.
136
142
  */
137
143
  export type GeminiInteractionsCustomEvent =
138
144
  | GeminiInteractionIdEvent
@@ -21,6 +21,5 @@ export type ExternalTextInteractionsProviderOptions = Pick<
21
21
  | 'system_instruction'
22
22
  | 'response_modalities'
23
23
  | 'response_format'
24
- | 'response_mime_type'
25
24
  | 'generation_config'
26
25
  >
@@ -1,92 +0,0 @@
1
- import { BaseAudioAdapter } from '@tanstack/ai/adapters';
2
- import { GEMINI_AUDIO_MODELS } from '../model-meta.js';
3
- import { AudioGenerationOptions, AudioGenerationResult } from '@tanstack/ai';
4
- import { GeminiClientConfig } from '../utils.js';
5
- /**
6
- * Provider options for Gemini Lyria music generation.
7
- *
8
- * Notes on the Lyria 3 surface area:
9
- * - `lyria-3-clip-preview` always returns MP3 (30-second clips). It does
10
- * not accept `responseMimeType`, and duration is fixed at 30 seconds —
11
- * the generic `duration` option on `AudioActivityOptions` is ignored.
12
- * - `lyria-3-pro-preview` returns MP3 by default. Duration is controlled
13
- * via the natural-language prompt, not a separate SDK field, so the
14
- * generic `duration` option is similarly ignored.
15
- * - `negativePrompt` is NOT accepted by `GenerateContentConfig` and has
16
- * therefore been removed from this surface to avoid giving callers a
17
- * silently-dropped knob.
18
- *
19
- * @see https://ai.google.dev/gemini-api/docs/music-generation
20
- */
21
- export interface GeminiAudioProviderOptions {
22
- /**
23
- * Seed for deterministic generation.
24
- */
25
- seed?: number;
26
- }
27
- export interface GeminiAudioConfig extends GeminiClientConfig {
28
- }
29
- /** Model type for Gemini Lyria audio generation */
30
- export type GeminiAudioModel = (typeof GEMINI_AUDIO_MODELS)[number];
31
- /**
32
- * Gemini Lyria Music Generation Adapter.
33
- *
34
- * Tree-shakeable adapter for Google Lyria music generation via the Gemini API.
35
- *
36
- * Models:
37
- * - `lyria-3-pro-preview` — flagship model, full-length songs with verses,
38
- * choruses, and bridges. Outputs MP3 or WAV at 48 kHz stereo.
39
- * - `lyria-3-clip-preview` — 30-second clips in MP3.
40
- *
41
- * @see https://ai.google.dev/gemini-api/docs/music-generation
42
- *
43
- * @example
44
- * ```typescript
45
- * const adapter = geminiAudio('lyria-3-pro-preview')
46
- * const result = await generateAudio({
47
- * adapter,
48
- * prompt: 'An upbeat jazz track with saxophone and drums',
49
- * })
50
- * ```
51
- */
52
- export declare class GeminiAudioAdapter<TModel extends GeminiAudioModel> extends BaseAudioAdapter<TModel, GeminiAudioProviderOptions> {
53
- readonly name: "gemini";
54
- private readonly client;
55
- constructor(config: GeminiAudioConfig, model: TModel);
56
- generateAudio(options: AudioGenerationOptions<GeminiAudioProviderOptions>): Promise<AudioGenerationResult>;
57
- }
58
- /**
59
- * Creates a Gemini Lyria audio adapter with an explicit API key.
60
- *
61
- * @param model - The Lyria model name (e.g., 'lyria-3-pro-preview')
62
- * @param apiKey - Your Google API key
63
- * @param config - Optional additional configuration
64
- *
65
- * @example
66
- * ```typescript
67
- * const adapter = createGeminiAudio('lyria-3-pro-preview', 'your-api-key')
68
- * const result = await generateAudio({
69
- * adapter,
70
- * prompt: 'Ambient electronic music with soft pads',
71
- * })
72
- * ```
73
- */
74
- export declare function createGeminiAudio<TModel extends GeminiAudioModel>(model: TModel, apiKey: string, config?: Omit<GeminiAudioConfig, 'apiKey'>): GeminiAudioAdapter<TModel>;
75
- /**
76
- * Creates a Gemini Lyria audio adapter with automatic API key detection.
77
- *
78
- * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in the environment.
79
- *
80
- * @param model - The Lyria model name (e.g., 'lyria-3-pro-preview')
81
- * @param config - Optional configuration (excluding apiKey)
82
- *
83
- * @example
84
- * ```typescript
85
- * const adapter = geminiAudio('lyria-3-pro-preview')
86
- * const result = await generateAudio({
87
- * adapter,
88
- * prompt: 'An orchestral piece with strings and brass',
89
- * })
90
- * ```
91
- */
92
- export declare function geminiAudio<TModel extends GeminiAudioModel>(model: TModel, config?: Omit<GeminiAudioConfig, 'apiKey'>): GeminiAudioAdapter<TModel>;
@@ -1,66 +0,0 @@
1
- import { BaseAudioAdapter } from "@tanstack/ai/adapters";
2
- import { createGeminiClient, generateId, getGeminiApiKeyFromEnv } from "../utils/client.js";
3
- import { buildGeminiUsage } from "../usage.js";
4
- class GeminiAudioAdapter extends BaseAudioAdapter {
5
- name = "gemini";
6
- client;
7
- constructor(config, model) {
8
- super(model, config);
9
- this.client = createGeminiClient(config);
10
- }
11
- async generateAudio(options) {
12
- const { model, prompt, modelOptions, logger } = options;
13
- logger.request(`activity=generateAudio provider=gemini model=${model}`, {
14
- provider: "gemini",
15
- model
16
- });
17
- try {
18
- const response = await this.client.models.generateContent({
19
- model,
20
- contents: [{ role: "user", parts: [{ text: prompt }] }],
21
- config: {
22
- responseModalities: ["AUDIO", "TEXT"],
23
- ...modelOptions?.seed != null ? { seed: modelOptions.seed } : {}
24
- }
25
- });
26
- const parts = response.candidates?.[0]?.content?.parts ?? [];
27
- const audioPart = parts.find(
28
- (part) => part.inlineData?.mimeType?.startsWith("audio/")
29
- );
30
- if (!audioPart?.inlineData?.data) {
31
- throw new Error("No audio data in Gemini Lyria response");
32
- }
33
- const contentType = audioPart.inlineData.mimeType;
34
- return {
35
- id: generateId(this.name),
36
- model,
37
- audio: {
38
- b64Json: audioPart.inlineData.data,
39
- ...contentType !== void 0 && { contentType }
40
- },
41
- // Surface token usage (with per-modality breakdown) when Gemini reports
42
- // it. Spread conditionally for exactOptionalPropertyTypes.
43
- ...response.usageMetadata ? { usage: buildGeminiUsage(response.usageMetadata) } : {}
44
- };
45
- } catch (error) {
46
- logger.errors("gemini.generateAudio fatal", {
47
- error,
48
- source: "gemini.generateAudio"
49
- });
50
- throw error;
51
- }
52
- }
53
- }
54
- function createGeminiAudio(model, apiKey, config) {
55
- return new GeminiAudioAdapter({ ...config, apiKey }, model);
56
- }
57
- function geminiAudio(model, config) {
58
- const apiKey = getGeminiApiKeyFromEnv();
59
- return createGeminiAudio(model, apiKey, config);
60
- }
61
- export {
62
- GeminiAudioAdapter,
63
- createGeminiAudio,
64
- geminiAudio
65
- };
66
- //# sourceMappingURL=audio.js.map
@@ -1 +0,0 @@
1
- {"version":3,"file":"audio.js","sources":["../../../src/adapters/audio.ts"],"sourcesContent":["import { BaseAudioAdapter } from '@tanstack/ai/adapters'\nimport {\n createGeminiClient,\n generateId,\n getGeminiApiKeyFromEnv,\n} from '../utils'\nimport { buildGeminiUsage } from '../usage'\nimport type { GEMINI_AUDIO_MODELS } from '../model-meta'\nimport type {\n AudioGenerationOptions,\n AudioGenerationResult,\n} from '@tanstack/ai'\nimport type { GoogleGenAI } from '@google/genai'\nimport type { GeminiClientConfig } from '../utils'\n\n/**\n * Provider options for Gemini Lyria music generation.\n *\n * Notes on the Lyria 3 surface area:\n * - `lyria-3-clip-preview` always returns MP3 (30-second clips). It does\n * not accept `responseMimeType`, and duration is fixed at 30 seconds —\n * the generic `duration` option on `AudioActivityOptions` is ignored.\n * - `lyria-3-pro-preview` returns MP3 by default. Duration is controlled\n * via the natural-language prompt, not a separate SDK field, so the\n * generic `duration` option is similarly ignored.\n * - `negativePrompt` is NOT accepted by `GenerateContentConfig` and has\n * therefore been removed from this surface to avoid giving callers a\n * silently-dropped knob.\n *\n * @see https://ai.google.dev/gemini-api/docs/music-generation\n */\nexport interface GeminiAudioProviderOptions {\n /**\n * Seed for deterministic generation.\n */\n seed?: number\n}\n\nexport interface GeminiAudioConfig extends GeminiClientConfig {}\n\n/** Model type for Gemini Lyria audio generation */\nexport type GeminiAudioModel = (typeof GEMINI_AUDIO_MODELS)[number]\n\n/**\n * Gemini Lyria Music Generation Adapter.\n *\n * Tree-shakeable adapter for Google Lyria music generation via the Gemini API.\n *\n * Models:\n * - `lyria-3-pro-preview` — flagship model, full-length songs with verses,\n * choruses, and bridges. Outputs MP3 or WAV at 48 kHz stereo.\n * - `lyria-3-clip-preview` — 30-second clips in MP3.\n *\n * @see https://ai.google.dev/gemini-api/docs/music-generation\n *\n * @example\n * ```typescript\n * const adapter = geminiAudio('lyria-3-pro-preview')\n * const result = await generateAudio({\n * adapter,\n * prompt: 'An upbeat jazz track with saxophone and drums',\n * })\n * ```\n */\nexport class GeminiAudioAdapter<\n TModel extends GeminiAudioModel,\n> extends BaseAudioAdapter<TModel, GeminiAudioProviderOptions> {\n readonly name = 'gemini' as const\n\n private readonly client: GoogleGenAI\n\n constructor(config: GeminiAudioConfig, model: TModel) {\n super(model, config)\n this.client = createGeminiClient(config)\n }\n\n async generateAudio(\n options: AudioGenerationOptions<GeminiAudioProviderOptions>,\n ): Promise<AudioGenerationResult> {\n const { model, prompt, modelOptions, logger } = options\n\n logger.request(`activity=generateAudio provider=gemini model=${model}`, {\n provider: 'gemini',\n model,\n })\n\n try {\n // FIXME (SDK audit): Lyria 3 music generation may not belong on\n // generateContent at all — @google/genai exposes a `LiveMusicSession`\n // (`ai.live.music.connect`) with a `musicGenerationConfig` object.\n // `seed` is valid on GenerateContentConfig, and Lyria always returns\n // MP3 today, so we don't forward `responseMimeType` either.\n // The runtime test `emits only GenerateContentConfig-valid fields`\n // asserts the config shape so a later SDK audit can catch regressions.\n const response = await this.client.models.generateContent({\n model,\n contents: [{ role: 'user', parts: [{ text: prompt }] }],\n config: {\n responseModalities: ['AUDIO', 'TEXT'],\n ...(modelOptions?.seed != null ? { seed: modelOptions.seed } : {}),\n },\n })\n\n const parts = response.candidates?.[0]?.content?.parts ?? []\n const audioPart = parts.find((part: any) =>\n part.inlineData?.mimeType?.startsWith('audio/'),\n )\n\n if (!audioPart?.inlineData?.data) {\n throw new Error('No audio data in Gemini Lyria response')\n }\n\n // audioPart was selected because mimeType.startsWith('audio/') was\n // truthy, so the mime type is guaranteed to be a string here. Trust the\n // value Gemini returned rather than inventing a non-standard\n // `audio/mp3` fallback (IANA is `audio/mpeg`).\n const contentType = audioPart.inlineData.mimeType\n\n return {\n id: generateId(this.name),\n model,\n audio: {\n b64Json: audioPart.inlineData.data,\n ...(contentType !== undefined && { contentType }),\n },\n // Surface token usage (with per-modality breakdown) when Gemini reports\n // it. Spread conditionally for exactOptionalPropertyTypes.\n ...(response.usageMetadata\n ? { usage: buildGeminiUsage(response.usageMetadata) }\n : {}),\n }\n } catch (error) {\n logger.errors('gemini.generateAudio fatal', {\n error,\n source: 'gemini.generateAudio',\n })\n throw error\n }\n }\n}\n\n/**\n * Creates a Gemini Lyria audio adapter with an explicit API key.\n *\n * @param model - The Lyria model name (e.g., 'lyria-3-pro-preview')\n * @param apiKey - Your Google API key\n * @param config - Optional additional configuration\n *\n * @example\n * ```typescript\n * const adapter = createGeminiAudio('lyria-3-pro-preview', 'your-api-key')\n * const result = await generateAudio({\n * adapter,\n * prompt: 'Ambient electronic music with soft pads',\n * })\n * ```\n */\nexport function createGeminiAudio<TModel extends GeminiAudioModel>(\n model: TModel,\n apiKey: string,\n config?: Omit<GeminiAudioConfig, 'apiKey'>,\n): GeminiAudioAdapter<TModel> {\n // Put apiKey LAST so caller-supplied config can't silently override the\n // explicit argument.\n return new GeminiAudioAdapter({ ...config, apiKey }, model)\n}\n\n/**\n * Creates a Gemini Lyria audio adapter with automatic API key detection.\n *\n * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in the environment.\n *\n * @param model - The Lyria model name (e.g., 'lyria-3-pro-preview')\n * @param config - Optional configuration (excluding apiKey)\n *\n * @example\n * ```typescript\n * const adapter = geminiAudio('lyria-3-pro-preview')\n * const result = await generateAudio({\n * adapter,\n * prompt: 'An orchestral piece with strings and brass',\n * })\n * ```\n */\nexport function geminiAudio<TModel extends GeminiAudioModel>(\n model: TModel,\n config?: Omit<GeminiAudioConfig, 'apiKey'>,\n): GeminiAudioAdapter<TModel> {\n const apiKey = getGeminiApiKeyFromEnv()\n return createGeminiAudio(model, apiKey, config)\n}\n"],"names":[],"mappings":";;;AAgEO,MAAM,2BAEH,iBAAqD;AAAA,EACpD,OAAO;AAAA,EAEC;AAAA,EAEjB,YAAY,QAA2B,OAAe;AACpD,UAAM,OAAO,MAAM;AACnB,SAAK,SAAS,mBAAmB,MAAM;AAAA,EACzC;AAAA,EAEA,MAAM,cACJ,SACgC;AAChC,UAAM,EAAE,OAAO,QAAQ,cAAc,WAAW;AAEhD,WAAO,QAAQ,gDAAgD,KAAK,IAAI;AAAA,MACtE,UAAU;AAAA,MACV;AAAA,IAAA,CACD;AAED,QAAI;AAQF,YAAM,WAAW,MAAM,KAAK,OAAO,OAAO,gBAAgB;AAAA,QACxD;AAAA,QACA,UAAU,CAAC,EAAE,MAAM,QAAQ,OAAO,CAAC,EAAE,MAAM,OAAA,CAAQ,GAAG;AAAA,QACtD,QAAQ;AAAA,UACN,oBAAoB,CAAC,SAAS,MAAM;AAAA,UACpC,GAAI,cAAc,QAAQ,OAAO,EAAE,MAAM,aAAa,SAAS,CAAA;AAAA,QAAC;AAAA,MAClE,CACD;AAED,YAAM,QAAQ,SAAS,aAAa,CAAC,GAAG,SAAS,SAAS,CAAA;AAC1D,YAAM,YAAY,MAAM;AAAA,QAAK,CAAC,SAC5B,KAAK,YAAY,UAAU,WAAW,QAAQ;AAAA,MAAA;AAGhD,UAAI,CAAC,WAAW,YAAY,MAAM;AAChC,cAAM,IAAI,MAAM,wCAAwC;AAAA,MAC1D;AAMA,YAAM,cAAc,UAAU,WAAW;AAEzC,aAAO;AAAA,QACL,IAAI,WAAW,KAAK,IAAI;AAAA,QACxB;AAAA,QACA,OAAO;AAAA,UACL,SAAS,UAAU,WAAW;AAAA,UAC9B,GAAI,gBAAgB,UAAa,EAAE,YAAA;AAAA,QAAY;AAAA;AAAA;AAAA,QAIjD,GAAI,SAAS,gBACT,EAAE,OAAO,iBAAiB,SAAS,aAAa,MAChD,CAAA;AAAA,MAAC;AAAA,IAET,SAAS,OAAO;AACd,aAAO,OAAO,8BAA8B;AAAA,QAC1C;AAAA,QACA,QAAQ;AAAA,MAAA,CACT;AACD,YAAM;AAAA,IACR;AAAA,EACF;AACF;AAkBO,SAAS,kBACd,OACA,QACA,QAC4B;AAG5B,SAAO,IAAI,mBAAmB,EAAE,GAAG,QAAQ,OAAA,GAAU,KAAK;AAC5D;AAmBO,SAAS,YACd,OACA,QAC4B;AAC5B,QAAM,SAAS,uBAAA;AACf,SAAO,kBAAkB,OAAO,QAAQ,MAAM;AAChD;"}
@@ -1,101 +0,0 @@
1
- import { BaseImageAdapter } from '@tanstack/ai/adapters';
2
- import { GEMINI_IMAGE_MODELS } from '../model-meta.js';
3
- import { GeminiImageModelInputModalitiesByName, GeminiImageModelProviderOptionsByName, GeminiImageModelSizeByName, GeminiImageProviderOptions } from '../image/image-provider-options.js';
4
- import { ImageGenerationOptions, ImageGenerationResult } from '@tanstack/ai';
5
- import { GeminiClientConfig } from '../utils.js';
6
- /**
7
- * Configuration for Gemini image adapter
8
- */
9
- export interface GeminiImageConfig extends GeminiClientConfig {
10
- }
11
- /** Model type for Gemini Image */
12
- export type GeminiImageModel = (typeof GEMINI_IMAGE_MODELS)[number];
13
- /**
14
- * Gemini Image Generation Adapter
15
- *
16
- * Tree-shakeable adapter for Gemini image generation functionality.
17
- * Supports Imagen 3/4 models (via generateImages API) and Gemini native
18
- * image models like Nano Banana 2 (via generateContent API).
19
- *
20
- * Features:
21
- * - Aspect ratio-based image sizing
22
- * - Person generation controls
23
- * - Safety filtering
24
- * - Watermark options
25
- * - Extended resolution tiers (Nano Banana 2)
26
- */
27
- export declare class GeminiImageAdapter<TModel extends GeminiImageModel> extends BaseImageAdapter<TModel, GeminiImageProviderOptions, GeminiImageModelProviderOptionsByName, GeminiImageModelSizeByName, GeminiImageModelInputModalitiesByName> {
28
- readonly kind: "image";
29
- readonly name: "gemini";
30
- '~types': {
31
- providerOptions: GeminiImageProviderOptions;
32
- modelProviderOptionsByName: GeminiImageModelProviderOptionsByName;
33
- modelSizeByName: GeminiImageModelSizeByName;
34
- modelInputModalitiesByName: GeminiImageModelInputModalitiesByName;
35
- };
36
- private readonly client;
37
- constructor(config: GeminiImageConfig, model: TModel);
38
- generateImages(options: ImageGenerationOptions<GeminiImageProviderOptions>): Promise<ImageGenerationResult>;
39
- private isGeminiImageModel;
40
- private generateWithGeminiApi;
41
- /**
42
- * Build the multimodal `contents` payload. Text-only prompts pass through
43
- * as a plain string (the SDK accepts it directly); prompts with image
44
- * parts become a single user `Content` whose `parts` mirror the prompt's
45
- * interleaved order — position is meaningful to Gemini ("not like this
46
- * *(image)*, more like this *(image)*").
47
- *
48
- * The generateContent API has no numberOfImages parameter, so when more
49
- * than one image is requested a trailing instruction is appended.
50
- */
51
- private buildContents;
52
- private imagePartToGeminiPart;
53
- private transformGeminiResponse;
54
- private buildImagenConfig;
55
- private transformImagenResponse;
56
- }
57
- /**
58
- * Creates a Gemini image adapter with explicit API key.
59
- * Type resolution happens here at the call site.
60
- *
61
- * @param model - The model name (e.g., 'imagen-3.0-generate-002')
62
- * @param apiKey - Your Google API key
63
- * @param config - Optional additional configuration
64
- * @returns Configured Gemini image adapter instance with resolved types
65
- *
66
- * @example
67
- * ```typescript
68
- * const adapter = createGeminiImage('imagen-3.0-generate-002', "your-api-key");
69
- *
70
- * const result = await generateImage({
71
- * adapter,
72
- * prompt: 'A cute baby sea otter'
73
- * });
74
- * ```
75
- */
76
- export declare function createGeminiImage<TModel extends GeminiImageModel>(model: TModel, apiKey: string, config?: Omit<GeminiImageConfig, 'apiKey'>): GeminiImageAdapter<TModel>;
77
- /**
78
- * Creates a Gemini image adapter with automatic API key detection from environment variables.
79
- * Type resolution happens here at the call site.
80
- *
81
- * Looks for `GOOGLE_API_KEY` or `GEMINI_API_KEY` in:
82
- * - `process.env` (Node.js)
83
- * - `window.env` (Browser with injected env)
84
- *
85
- * @param model - The model name (e.g., 'imagen-4.0-generate-001')
86
- * @param config - Optional configuration (excluding apiKey which is auto-detected)
87
- * @returns Configured Gemini image adapter instance with resolved types
88
- * @throws Error if GOOGLE_API_KEY or GEMINI_API_KEY is not found in environment
89
- *
90
- * @example
91
- * ```typescript
92
- * // Automatically uses GOOGLE_API_KEY from environment
93
- * const adapter = geminiImage('imagen-4.0-generate-001');
94
- *
95
- * const result = await generateImage({
96
- * adapter,
97
- * prompt: 'A beautiful sunset over mountains'
98
- * });
99
- * ```
100
- */
101
- export declare function geminiImage<TModel extends GeminiImageModel>(model: TModel, config?: Omit<GeminiImageConfig, 'apiKey'>): GeminiImageAdapter<TModel>;
@@ -1,253 +0,0 @@
1
- import { resolveMediaPrompt } from "@tanstack/ai";
2
- import { BaseImageAdapter } from "@tanstack/ai/adapters";
3
- import { arrayBufferToBase64 } from "@tanstack/ai-utils";
4
- import { createGeminiClient, generateId, getGeminiApiKeyFromEnv } from "../utils/client.js";
5
- import { buildGeminiUsage } from "../usage.js";
6
- import { validatePrompt, validateImageSize, validateNumberOfImages, parseNativeImageSize, sizeToAspectRatio } from "../image/image-provider-options.js";
7
- class GeminiImageAdapter extends BaseImageAdapter {
8
- kind = "image";
9
- name = "gemini";
10
- client;
11
- constructor(config, model) {
12
- super(model, config);
13
- this.client = createGeminiClient(config);
14
- }
15
- async generateImages(options) {
16
- const { model, logger } = options;
17
- logger.request(
18
- `activity=generateImage provider=gemini model=${this.model}`,
19
- {
20
- provider: "gemini",
21
- model: this.model
22
- }
23
- );
24
- try {
25
- const resolved = resolveMediaPrompt(options.prompt);
26
- if (resolved.images.length === 0) {
27
- validatePrompt({ prompt: resolved.text, model });
28
- }
29
- if (resolved.videos.length > 0) {
30
- throw new Error(
31
- `${this.name}.generateImages does not support video prompt parts (model: ${model}).`
32
- );
33
- }
34
- if (resolved.audios.length > 0) {
35
- throw new Error(
36
- `${this.name}.generateImages does not support audio prompt parts (model: ${model}).`
37
- );
38
- }
39
- if (this.isGeminiImageModel(model)) {
40
- return await this.generateWithGeminiApi(options, resolved);
41
- }
42
- if (resolved.images.length > 0) {
43
- throw new Error(
44
- `${this.name}: model "${model}" (Imagen) does not support image prompt parts. Use a Gemini-native image model (e.g. gemini-2.5-flash-image, "nano-banana") for image-conditioned generation.`
45
- );
46
- }
47
- validateImageSize(model, options.size);
48
- validateNumberOfImages(model, options.numberOfImages);
49
- const config = this.buildImagenConfig(options);
50
- const response = await this.client.models.generateImages({
51
- model,
52
- prompt: resolved.text,
53
- config
54
- });
55
- return this.transformImagenResponse(model, response);
56
- } catch (error) {
57
- logger.errors("gemini.generateImage fatal", {
58
- error,
59
- source: "gemini.generateImage"
60
- });
61
- throw error;
62
- }
63
- }
64
- isGeminiImageModel(model) {
65
- return model.startsWith("gemini-");
66
- }
67
- async generateWithGeminiApi(options, resolved) {
68
- const { model, size, numberOfImages, modelOptions } = options;
69
- const parsedSize = size ? parseNativeImageSize(size) : void 0;
70
- const nativeConfig = {};
71
- if (modelOptions?.seed !== void 0) {
72
- nativeConfig.seed = modelOptions.seed;
73
- }
74
- const config = {
75
- ...nativeConfig,
76
- // Include TEXT so the model can interleave descriptions between images.
77
- // IMPORTANT: responseModalities is a protected default — set it AFTER
78
- // nativeConfig so nothing can silently disable image output.
79
- responseModalities: ["TEXT", "IMAGE"],
80
- ...parsedSize && {
81
- imageConfig: {
82
- ...parsedSize.aspectRatio && {
83
- aspectRatio: parsedSize.aspectRatio
84
- },
85
- ...parsedSize.resolution && {
86
- imageSize: parsedSize.resolution
87
- }
88
- }
89
- }
90
- };
91
- const contents = await this.buildContents(resolved, numberOfImages);
92
- const response = await this.client.models.generateContent({
93
- model,
94
- contents,
95
- config
96
- });
97
- return this.transformGeminiResponse(model, response);
98
- }
99
- /**
100
- * Build the multimodal `contents` payload. Text-only prompts pass through
101
- * as a plain string (the SDK accepts it directly); prompts with image
102
- * parts become a single user `Content` whose `parts` mirror the prompt's
103
- * interleaved order — position is meaningful to Gemini ("not like this
104
- * *(image)*, more like this *(image)*").
105
- *
106
- * The generateContent API has no numberOfImages parameter, so when more
107
- * than one image is requested a trailing instruction is appended.
108
- */
109
- async buildContents(resolved, numberOfImages) {
110
- const countInstruction = numberOfImages && numberOfImages > 1 ? `Generate ${numberOfImages} distinct images.` : void 0;
111
- if (resolved.images.length === 0) {
112
- return countInstruction ? `${resolved.text} ${countInstruction}` : resolved.text;
113
- }
114
- const parts = await Promise.all(
115
- resolved.parts.map((part) => {
116
- if (part.type === "text") {
117
- return Promise.resolve({ text: part.content });
118
- }
119
- if (part.type === "image") {
120
- return this.imagePartToGeminiPart(part);
121
- }
122
- throw new Error(
123
- `gemini: unsupported prompt part type "${part.type}" in image generation.`
124
- );
125
- })
126
- );
127
- if (countInstruction) {
128
- parts.push({ text: countInstruction });
129
- }
130
- return [{ role: "user", parts }];
131
- }
132
- async imagePartToGeminiPart(part) {
133
- if (part.source.type === "data") {
134
- return {
135
- inlineData: {
136
- mimeType: part.source.mimeType || "image/png",
137
- data: part.source.value
138
- }
139
- };
140
- }
141
- if (part.source.value.startsWith("gs://") || /^https?:\/\/generativelanguage\.googleapis\.com\//.test(
142
- part.source.value
143
- )) {
144
- return {
145
- fileData: {
146
- fileUri: part.source.value,
147
- ...part.source.mimeType && { mimeType: part.source.mimeType }
148
- }
149
- };
150
- }
151
- const response = await fetch(part.source.value);
152
- if (!response.ok) {
153
- throw new Error(
154
- `Failed to fetch image input (${response.status} ${response.statusText}): ${part.source.value}`
155
- );
156
- }
157
- const blob = await response.blob();
158
- const buffer = await blob.arrayBuffer();
159
- const base64 = arrayBufferToBase64(buffer);
160
- return {
161
- inlineData: {
162
- mimeType: part.source.mimeType || blob.type || "image/png",
163
- data: base64
164
- }
165
- };
166
- }
167
- transformGeminiResponse(model, response) {
168
- const images = [];
169
- const textParts = [];
170
- const parts = response.candidates?.[0]?.content?.parts ?? [];
171
- for (const part of parts) {
172
- if (part.inlineData?.data && typeof part.inlineData.data === "string" && part.inlineData.data.length > 0) {
173
- images.push({ b64Json: part.inlineData.data });
174
- } else if (typeof part.text === "string" && part.text.length > 0) {
175
- textParts.push(part.text);
176
- }
177
- }
178
- if (images.length === 0) {
179
- const reason = textParts.length > 0 ? `: ${textParts.join(" ").trim()}` : " (no inline image or text parts were returned).";
180
- throw new Error(`Gemini ${model} returned no images${reason}`);
181
- }
182
- return {
183
- id: generateId(this.name),
184
- model,
185
- images,
186
- // Surface token usage (with per-modality breakdown) when the model
187
- // reports it (e.g. Nano Banana via generateContent). Conditionally spread
188
- // to satisfy exactOptionalPropertyTypes — only include usage when
189
- // present. See #330.
190
- ...response.usageMetadata ? { usage: buildGeminiUsage(response.usageMetadata) } : {}
191
- };
192
- }
193
- buildImagenConfig(options) {
194
- const { size, numberOfImages, modelOptions } = options;
195
- const sizeAspectRatio = size ? sizeToAspectRatio(size) : void 0;
196
- return {
197
- numberOfImages: numberOfImages ?? 1,
198
- // Map size to aspect ratio if provided (modelOptions.aspectRatio will override)
199
- ...sizeAspectRatio !== void 0 && { aspectRatio: sizeAspectRatio },
200
- ...modelOptions
201
- };
202
- }
203
- transformImagenResponse(model, response) {
204
- const entries = response.generatedImages ?? [];
205
- const images = [];
206
- const filterReasons = [];
207
- for (const item of entries) {
208
- const b64Json = item.image?.imageBytes;
209
- if (b64Json) {
210
- images.push({
211
- b64Json,
212
- ...item.enhancedPrompt !== void 0 && {
213
- revisedPrompt: item.enhancedPrompt
214
- }
215
- });
216
- continue;
217
- }
218
- const reason = item.raiFilteredReason;
219
- if (reason) {
220
- filterReasons.push(reason);
221
- }
222
- }
223
- if (entries.length > 0 && images.length === 0) {
224
- const joined = filterReasons.length > 0 ? filterReasons.join("; ") : "";
225
- throw new Error(
226
- `Imagen ${model} returned no images: all ${entries.length} generated image(s) were filtered by Responsible-AI${joined ? ` (${joined})` : ""}.`
227
- );
228
- }
229
- if (filterReasons.length > 0 && typeof console !== "undefined") {
230
- console.warn(
231
- `[gemini-image] ${filterReasons.length} of ${entries.length} images from ${model} were filtered by Responsible-AI: ${filterReasons.join("; ")}`
232
- );
233
- }
234
- return {
235
- id: generateId(this.name),
236
- model,
237
- images
238
- };
239
- }
240
- }
241
- function createGeminiImage(model, apiKey, config) {
242
- return new GeminiImageAdapter({ apiKey, ...config }, model);
243
- }
244
- function geminiImage(model, config) {
245
- const apiKey = getGeminiApiKeyFromEnv();
246
- return createGeminiImage(model, apiKey, config);
247
- }
248
- export {
249
- GeminiImageAdapter,
250
- createGeminiImage,
251
- geminiImage
252
- };
253
- //# sourceMappingURL=image.js.map