@tanstack/ai-gemini 0.23.0 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.d.ts +14 -7
- package/dist/esm/adapters/image.js +27 -56
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/video.js +2 -1
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/experimental/text-interactions/adapter.js +3 -9
- package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
- package/dist/esm/image/image-provider-options.d.ts +167 -20
- package/dist/esm/image/image-provider-options.js +52 -5
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +3 -1
- package/dist/esm/index.js +3 -1
- package/dist/esm/model-meta.d.ts +15 -2
- package/dist/esm/model-meta.js +82 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.js +1 -5
- package/dist/esm/realtime/adapter.js.map +1 -1
- package/dist/esm/realtime/client.js.map +1 -1
- package/dist/esm/realtime/token.js +3 -2
- package/dist/esm/realtime/token.js.map +1 -1
- package/dist/esm/realtime/utils.js +4 -1
- package/dist/esm/realtime/utils.js.map +1 -1
- package/dist/esm/tools/tool-converter.js +0 -1
- package/dist/esm/tools/tool-converter.js.map +1 -1
- package/dist/esm/usage.js +1 -3
- package/dist/esm/usage.js.map +1 -1
- package/package.json +5 -5
- package/src/adapters/image.ts +123 -36
- package/src/image/image-provider-options.ts +226 -25
- package/src/index.ts +28 -0
- package/src/model-meta.ts +118 -1
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { GeminiImageModels } from '../model-meta.js';
|
|
2
|
-
import { ImagePromptLanguage, PersonGeneration, SafetyFilterLevel } from '@google/genai';
|
|
3
|
-
export type { ImagePromptLanguage, PersonGeneration, SafetyFilterLevel };
|
|
2
|
+
import { ContentUnion, ImageConfig, ImagePromptLanguage, PersonGeneration, SafetyFilterLevel, SafetySetting, ThinkingConfig } from '@google/genai';
|
|
3
|
+
export type { ContentUnion, ImageConfig, ImagePromptLanguage, PersonGeneration, SafetyFilterLevel, SafetySetting, ThinkingConfig, };
|
|
4
4
|
/**
|
|
5
5
|
* Gemini Imagen aspect ratio options
|
|
6
6
|
* Controls the aspect ratio of generated images
|
|
@@ -91,11 +91,70 @@ export interface GeminiImageProviderOptions {
|
|
|
91
91
|
labels?: Record<string, string>;
|
|
92
92
|
}
|
|
93
93
|
/**
|
|
94
|
-
*
|
|
95
|
-
*
|
|
94
|
+
* Provider options for Gemini native image models (Nano Banana and friends).
|
|
95
|
+
*
|
|
96
|
+
* These models are served by `generateContent`, not `generateImages`, so they
|
|
97
|
+
* are configured by @google/genai's `GenerateContentConfig` — a different
|
|
98
|
+
* shape from the Imagen-only {@link GeminiImageProviderOptions} above. Only
|
|
99
|
+
* the `GenerateContentConfig` fields with clear image-generation semantics are
|
|
100
|
+
* surfaced; sampling knobs (`temperature`, `topK`, …) and chat-only plumbing
|
|
101
|
+
* (`tools`, `responseSchema`, …) are deliberately left out.
|
|
102
|
+
*
|
|
103
|
+
* `responseModalities` is intentionally absent: the adapter always requests
|
|
104
|
+
* `['TEXT', 'IMAGE']`, and letting a caller override it would silently disable
|
|
105
|
+
* image output on an image-generation call.
|
|
106
|
+
*/
|
|
107
|
+
export interface GeminiNativeImageProviderOptions {
|
|
108
|
+
/**
|
|
109
|
+
* Optional seed for reproducible image generation
|
|
110
|
+
* When the same seed is used with the same prompt and settings,
|
|
111
|
+
* you should get similar (though not identical) results
|
|
112
|
+
*/
|
|
113
|
+
seed?: number;
|
|
114
|
+
/**
|
|
115
|
+
* Per-category safety thresholds applied to the request
|
|
116
|
+
* Each entry pairs a HarmCategory with a HarmBlockThreshold
|
|
117
|
+
*/
|
|
118
|
+
safetySettings?: Array<SafetySetting>;
|
|
119
|
+
/**
|
|
120
|
+
* Controls the model's internal reasoning before it emits an image
|
|
121
|
+
* Use to raise or disable the thinking budget on models that support it
|
|
122
|
+
*/
|
|
123
|
+
thinkingConfig?: ThinkingConfig;
|
|
124
|
+
/**
|
|
125
|
+
* Native image output controls. Merged over the values derived from the
|
|
126
|
+
* portable `size` option, so fields set here win per field while the rest
|
|
127
|
+
* of `size` is preserved.
|
|
128
|
+
*
|
|
129
|
+
* Only `aspectRatio` and `imageSize` are accepted on the Gemini Developer
|
|
130
|
+
* API. Other SDK `ImageConfig` keys throw on this surface.
|
|
131
|
+
*/
|
|
132
|
+
imageConfig?: GeminiNativeImageConfig;
|
|
133
|
+
/**
|
|
134
|
+
* System-level instructions that steer the model for the whole request,
|
|
135
|
+
* e.g. a house art direction applied on top of the per-call prompt
|
|
136
|
+
*/
|
|
137
|
+
systemInstruction?: ContentUnion;
|
|
138
|
+
}
|
|
139
|
+
/**
|
|
140
|
+
* Every provider-option field this adapter understands, across both API
|
|
141
|
+
* paths. Used as the adapter's base (model-agnostic) option type; the
|
|
142
|
+
* per-model map below is what narrows a given model to the half that
|
|
143
|
+
* actually applies to it.
|
|
144
|
+
*/
|
|
145
|
+
export type GeminiAnyImageProviderOptions = GeminiImageProviderOptions & GeminiNativeImageProviderOptions;
|
|
146
|
+
/**
|
|
147
|
+
* Model-specific provider options mapping.
|
|
148
|
+
* Gemini native image models go through `generateContent` and take
|
|
149
|
+
* `GenerateContentConfig` fields; Imagen models go through `generateImages`
|
|
150
|
+
* and take `GenerateImagesConfig` fields. Mirrors the native/Imagen split in
|
|
151
|
+
* {@link GeminiImageModelSizeByName} and
|
|
152
|
+
* {@link GeminiImageModelInputModalitiesByName}.
|
|
96
153
|
*/
|
|
97
154
|
export type GeminiImageModelProviderOptionsByName = {
|
|
98
|
-
[K in
|
|
155
|
+
[K in GeminiNativeImageModels]: GeminiNativeImageProviderOptions;
|
|
156
|
+
} & {
|
|
157
|
+
[K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: GeminiImageProviderOptions;
|
|
99
158
|
};
|
|
100
159
|
/**
|
|
101
160
|
* Supported size strings for Gemini Imagen models
|
|
@@ -103,30 +162,111 @@ export type GeminiImageModelProviderOptionsByName = {
|
|
|
103
162
|
*/
|
|
104
163
|
export type GeminiImageSize = '1024x1024' | '512x512' | '1024x768' | '1536x1024' | '1792x1024' | '1920x1080' | '768x1024' | '1024x1536' | '1024x1792' | '1080x1920';
|
|
105
164
|
/**
|
|
106
|
-
*
|
|
107
|
-
*
|
|
165
|
+
* The ten aspect ratios every Gemini native image model accepts.
|
|
166
|
+
*
|
|
167
|
+
* Note `9:21` is deliberately absent: it exists only on Vertex / Cloud and is
|
|
168
|
+
* rejected by the Gemini API (`generateContent`), which is the surface this
|
|
169
|
+
* adapter targets.
|
|
170
|
+
*
|
|
171
|
+
* @see https://ai.google.dev/gemini-api/docs/image-generation
|
|
172
|
+
*/
|
|
173
|
+
export type GeminiStandardImageAspectRatio = '1:1' | '2:3' | '3:2' | '3:4' | '4:3' | '4:5' | '5:4' | '9:16' | '16:9' | '21:9';
|
|
174
|
+
/**
|
|
175
|
+
* The ten standard ratios plus the four extreme banner/strip ratios that only
|
|
176
|
+
* the Gemini 3.1 Flash Image models accept — 14 values, matching the
|
|
177
|
+
* `generateContent` `ImageConfig.aspectRatio` field union.
|
|
178
|
+
*
|
|
179
|
+
* @see https://ai.google.dev/api/generate-content
|
|
180
|
+
*/
|
|
181
|
+
export type GeminiExtendedImageAspectRatio = GeminiStandardImageAspectRatio | '1:4' | '4:1' | '1:8' | '8:1';
|
|
182
|
+
/**
|
|
183
|
+
* Sizes for `gemini-3.1-flash-image` (and its shut-down `-preview` alias):
|
|
184
|
+
* all 14 aspect ratios at 512 / 1K / 2K / 4K. `512` is the wire token for the
|
|
185
|
+
* 0.5K tier — not `512px`, and the `K` is case-sensitive (`1k` is rejected).
|
|
186
|
+
*/
|
|
187
|
+
export type Gemini31FlashImageSize = `${GeminiExtendedImageAspectRatio}_${'512' | '1K' | '2K' | '4K'}`;
|
|
188
|
+
/**
|
|
189
|
+
* Sizes for `gemini-3.1-flash-lite-image`: all 14 aspect ratios, 1K only.
|
|
190
|
+
* 2K and 4K are unsupported on this model.
|
|
191
|
+
*
|
|
192
|
+
* The four banner ratios (`1:4`, `4:1`, `1:8`, `8:1`) come from the Cloud
|
|
193
|
+
* model page. The Gemini API page states a count of 14 but does not list them.
|
|
194
|
+
*
|
|
195
|
+
* @see https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/gemini/3-1-flash-lite-image
|
|
196
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image
|
|
197
|
+
*/
|
|
198
|
+
export type Gemini31FlashLiteImageSize = `${GeminiExtendedImageAspectRatio}_1K`;
|
|
199
|
+
/**
|
|
200
|
+
* Sizes for `gemini-3-pro-image` (and its shut-down `-preview` alias): the ten
|
|
201
|
+
* standard aspect ratios at 1K / 2K / 4K. Pro has no 512 tier and none of the
|
|
202
|
+
* extreme banner ratios on the Gemini API.
|
|
203
|
+
*/
|
|
204
|
+
export type Gemini3ProImageSize = `${GeminiStandardImageAspectRatio}_${'1K' | '2K' | '4K'}`;
|
|
205
|
+
/**
|
|
206
|
+
* Sizes for `gemini-2.5-flash-image`: a bare aspect ratio with no resolution
|
|
207
|
+
* suffix, e.g. `'16:9'`. Google documents no `image_size` value or default for
|
|
208
|
+
* this model — it emits a single fixed 1024px-class output — so the adapter
|
|
209
|
+
* sends `imageConfig.aspectRatio` and omits `imageSize` entirely rather than
|
|
210
|
+
* guessing a tier the API never documented.
|
|
108
211
|
*/
|
|
109
|
-
export type
|
|
212
|
+
export type Gemini25FlashImageSize = GeminiStandardImageAspectRatio;
|
|
110
213
|
/**
|
|
111
|
-
*
|
|
112
|
-
*
|
|
214
|
+
* `imageConfig` fields the Gemini Developer API accepts on `generateContent`.
|
|
215
|
+
* Other `@google/genai` `ImageConfig` keys (`personGeneration`,
|
|
216
|
+
* `outputMimeType`, and more) throw on this surface.
|
|
113
217
|
*/
|
|
114
|
-
export type
|
|
218
|
+
export type GeminiNativeImageConfig = {
|
|
219
|
+
aspectRatio?: GeminiExtendedImageAspectRatio;
|
|
220
|
+
imageSize?: '512' | '1K' | '2K' | '4K';
|
|
221
|
+
};
|
|
115
222
|
/**
|
|
116
|
-
*
|
|
223
|
+
* Any size accepted by any Gemini native image model. Prefer the per-model
|
|
224
|
+
* narrowing in {@link GeminiImageModelSizeByName} — this union is the widest
|
|
225
|
+
* possible set and accepts combinations no single model supports.
|
|
117
226
|
*/
|
|
118
|
-
export type GeminiNativeImageSize =
|
|
227
|
+
export type GeminiNativeImageSize = Gemini31FlashImageSize | Gemini31FlashLiteImageSize | Gemini3ProImageSize | Gemini25FlashImageSize;
|
|
119
228
|
/**
|
|
120
229
|
* Gemini native image models that use the generateContent API path.
|
|
121
|
-
* These models
|
|
230
|
+
* These models take an aspect-ratio-based size rather than Imagen's
|
|
231
|
+
* WIDTHxHEIGHT pixel strings.
|
|
232
|
+
*
|
|
233
|
+
* This array is the single source of truth for the native/Imagen split: the
|
|
234
|
+
* `GeminiNativeImageModels` union and the per-model option/size/modality maps
|
|
235
|
+
* all derive from it. The `satisfies` clause makes a typo (or a name that
|
|
236
|
+
* is not a known image model) a build error rather than a phantom key on every
|
|
237
|
+
* per-model map.
|
|
238
|
+
*
|
|
239
|
+
* It is also the single source of truth for the adapter's runtime routing
|
|
240
|
+
* — see {@link isGeminiNativeImageModel}. Adding a new `gemini-*` image model
|
|
241
|
+
* means adding it here as well as to `GEMINI_IMAGE_MODELS` in model-meta.
|
|
242
|
+
* Until it is listed here it routes to the Imagen API instead and fails
|
|
243
|
+
* loudly on the first call, rather than silently taking the wrong option
|
|
244
|
+
* shape.
|
|
245
|
+
*/
|
|
246
|
+
export declare const GEMINI_NATIVE_IMAGE_MODELS: readonly ["gemini-3.1-flash-image", "gemini-3.1-flash-image-preview", "gemini-3.1-flash-lite-image", "gemini-3-pro-image", "gemini-3-pro-image-preview", "gemini-2.5-flash-image"];
|
|
247
|
+
export type GeminiNativeImageModels = (typeof GEMINI_NATIVE_IMAGE_MODELS)[number];
|
|
248
|
+
/**
|
|
249
|
+
* Runtime counterpart to {@link GeminiNativeImageModels} — decides which of
|
|
250
|
+
* the two Gemini image APIs a model goes to.
|
|
251
|
+
*
|
|
252
|
+
* Membership in {@link GEMINI_NATIVE_IMAGE_MODELS}, not a `gemini-` prefix
|
|
253
|
+
* test, so the runtime route and the type-level split cannot drift apart. An
|
|
254
|
+
* id this package does not know about reaches the Imagen endpoint and fails
|
|
255
|
+
* there, which is the intended signal to add the model here rather than to
|
|
256
|
+
* have it silently take the native path with Imagen-shaped option types.
|
|
122
257
|
*/
|
|
123
|
-
export
|
|
258
|
+
export declare function isGeminiNativeImageModel(model: string): boolean;
|
|
124
259
|
/**
|
|
125
|
-
* Model-specific size options mapping.
|
|
126
|
-
*
|
|
260
|
+
* Model-specific size options mapping. Each native model gets its own ratio ×
|
|
261
|
+
* resolution set (they genuinely differ); Imagen models use pixel sizes.
|
|
127
262
|
*/
|
|
128
263
|
export type GeminiImageModelSizeByName = {
|
|
129
|
-
|
|
264
|
+
'gemini-3.1-flash-image': Gemini31FlashImageSize;
|
|
265
|
+
'gemini-3.1-flash-image-preview': Gemini31FlashImageSize;
|
|
266
|
+
'gemini-3.1-flash-lite-image': Gemini31FlashLiteImageSize;
|
|
267
|
+
'gemini-3-pro-image': Gemini3ProImageSize;
|
|
268
|
+
'gemini-3-pro-image-preview': Gemini3ProImageSize;
|
|
269
|
+
'gemini-2.5-flash-image': Gemini25FlashImageSize;
|
|
130
270
|
} & {
|
|
131
271
|
[K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: GeminiImageSize;
|
|
132
272
|
};
|
|
@@ -173,9 +313,16 @@ export declare function validatePrompt(options: {
|
|
|
173
313
|
}): void;
|
|
174
314
|
/**
|
|
175
315
|
* Parses a Gemini native image size string into its components.
|
|
176
|
-
*
|
|
316
|
+
*
|
|
317
|
+
* Format: `"aspectRatio_resolution"`, e.g. `"16:9_4K"` →
|
|
318
|
+
* `{ aspectRatio: "16:9", resolution: "4K" }`.
|
|
319
|
+
*
|
|
320
|
+
* The resolution suffix is optional: `gemini-2.5-flash-image` takes a bare
|
|
321
|
+
* aspect ratio (`"16:9"` → `{ aspectRatio: "16:9" }`) because Google documents
|
|
322
|
+
* no `image_size` for it, and the caller must then omit `imageSize` from the
|
|
323
|
+
* request rather than substituting a default.
|
|
177
324
|
*/
|
|
178
325
|
export declare function parseNativeImageSize(size: string): {
|
|
179
326
|
aspectRatio: string;
|
|
180
|
-
resolution
|
|
327
|
+
resolution?: string;
|
|
181
328
|
} | undefined;
|
|
@@ -1,5 +1,45 @@
|
|
|
1
1
|
//#region src/image/image-provider-options.ts
|
|
2
2
|
/**
|
|
3
|
+
* Gemini native image models that use the generateContent API path.
|
|
4
|
+
* These models take an aspect-ratio-based size rather than Imagen's
|
|
5
|
+
* WIDTHxHEIGHT pixel strings.
|
|
6
|
+
*
|
|
7
|
+
* This array is the single source of truth for the native/Imagen split: the
|
|
8
|
+
* `GeminiNativeImageModels` union and the per-model option/size/modality maps
|
|
9
|
+
* all derive from it. The `satisfies` clause makes a typo (or a name that
|
|
10
|
+
* is not a known image model) a build error rather than a phantom key on every
|
|
11
|
+
* per-model map.
|
|
12
|
+
*
|
|
13
|
+
* It is also the single source of truth for the adapter's runtime routing
|
|
14
|
+
* — see {@link isGeminiNativeImageModel}. Adding a new `gemini-*` image model
|
|
15
|
+
* means adding it here as well as to `GEMINI_IMAGE_MODELS` in model-meta.
|
|
16
|
+
* Until it is listed here it routes to the Imagen API instead and fails
|
|
17
|
+
* loudly on the first call, rather than silently taking the wrong option
|
|
18
|
+
* shape.
|
|
19
|
+
*/
|
|
20
|
+
var GEMINI_NATIVE_IMAGE_MODELS = [
|
|
21
|
+
"gemini-3.1-flash-image",
|
|
22
|
+
"gemini-3.1-flash-image-preview",
|
|
23
|
+
"gemini-3.1-flash-lite-image",
|
|
24
|
+
"gemini-3-pro-image",
|
|
25
|
+
"gemini-3-pro-image-preview",
|
|
26
|
+
"gemini-2.5-flash-image"
|
|
27
|
+
];
|
|
28
|
+
var NATIVE_IMAGE_MODEL_NAMES = new Set(GEMINI_NATIVE_IMAGE_MODELS);
|
|
29
|
+
/**
|
|
30
|
+
* Runtime counterpart to {@link GeminiNativeImageModels} — decides which of
|
|
31
|
+
* the two Gemini image APIs a model goes to.
|
|
32
|
+
*
|
|
33
|
+
* Membership in {@link GEMINI_NATIVE_IMAGE_MODELS}, not a `gemini-` prefix
|
|
34
|
+
* test, so the runtime route and the type-level split cannot drift apart. An
|
|
35
|
+
* id this package does not know about reaches the Imagen endpoint and fails
|
|
36
|
+
* there, which is the intended signal to add the model here rather than to
|
|
37
|
+
* have it silently take the native path with Imagen-shaped option types.
|
|
38
|
+
*/
|
|
39
|
+
function isGeminiNativeImageModel(model) {
|
|
40
|
+
return NATIVE_IMAGE_MODEL_NAMES.has(model);
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
3
43
|
* Valid sizes for Gemini Imagen models
|
|
4
44
|
* Gemini uses aspect ratios, but we map common WIDTHxHEIGHT formats to aspect ratios
|
|
5
45
|
* These are approximate mappings based on common image dimensions
|
|
@@ -70,17 +110,24 @@ function validatePrompt(options) {
|
|
|
70
110
|
}
|
|
71
111
|
/**
|
|
72
112
|
* Parses a Gemini native image size string into its components.
|
|
73
|
-
*
|
|
113
|
+
*
|
|
114
|
+
* Format: `"aspectRatio_resolution"`, e.g. `"16:9_4K"` →
|
|
115
|
+
* `{ aspectRatio: "16:9", resolution: "4K" }`.
|
|
116
|
+
*
|
|
117
|
+
* The resolution suffix is optional: `gemini-2.5-flash-image` takes a bare
|
|
118
|
+
* aspect ratio (`"16:9"` → `{ aspectRatio: "16:9" }`) because Google documents
|
|
119
|
+
* no `image_size` for it, and the caller must then omit `imageSize` from the
|
|
120
|
+
* request rather than substituting a default.
|
|
74
121
|
*/
|
|
75
122
|
function parseNativeImageSize(size) {
|
|
76
|
-
const [, aspectRatio, resolution] = size.match(/^(\d+:\d+)_(.+)
|
|
77
|
-
if (aspectRatio === void 0
|
|
123
|
+
const [, aspectRatio, resolution] = size.match(/^(\d+:\d+)(?:_(.+))?$/) ?? [];
|
|
124
|
+
if (aspectRatio === void 0) return void 0;
|
|
78
125
|
return {
|
|
79
126
|
aspectRatio,
|
|
80
|
-
resolution
|
|
127
|
+
...resolution !== void 0 && { resolution }
|
|
81
128
|
};
|
|
82
129
|
}
|
|
83
130
|
//#endregion
|
|
84
|
-
export { GEMINI_SIZE_TO_ASPECT_RATIO, parseNativeImageSize, sizeToAspectRatio, validateImageSize, validateNumberOfImages, validatePrompt };
|
|
131
|
+
export { GEMINI_NATIVE_IMAGE_MODELS, GEMINI_SIZE_TO_ASPECT_RATIO, isGeminiNativeImageModel, parseNativeImageSize, sizeToAspectRatio, validateImageSize, validateNumberOfImages, validatePrompt };
|
|
85
132
|
|
|
86
133
|
//# sourceMappingURL=image-provider-options.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"image-provider-options.js","names":[],"sources":["../../../src/image/image-provider-options.ts"],"sourcesContent":["import type { GeminiImageModels } from '../model-meta'\nimport type {\n ImagePromptLanguage,\n PersonGeneration,\n SafetyFilterLevel,\n} from '@google/genai'\n\n// Re-export SDK types so users can use them directly\nexport type { ImagePromptLanguage, PersonGeneration, SafetyFilterLevel }\n\n/**\n * Gemini Imagen aspect ratio options\n * Controls the aspect ratio of generated images\n */\nexport type GeminiAspectRatio =\n | '1:1'\n | '3:4'\n | '4:3'\n | '9:16'\n | '16:9'\n | '9:21'\n | '21:9'\n\n/**\n * Provider options for Gemini image generation\n * These options match the @google/genai GenerateImagesConfig interface\n * and can be spread directly into the API request.\n */\nexport interface GeminiImageProviderOptions {\n /**\n * The aspect ratio of generated images\n * @default '1:1'\n */\n aspectRatio?: GeminiAspectRatio\n\n /**\n * Controls whether people can appear in generated images\n * Use PersonGeneration enum values: DONT_ALLOW, ALLOW_ADULT, ALLOW_ALL\n * @default 'ALLOW_ADULT'\n */\n personGeneration?: PersonGeneration\n\n /**\n * Safety filter level for content filtering\n * Use SafetyFilterLevel enum values\n */\n safetyFilterLevel?: SafetyFilterLevel\n\n /**\n * Optional seed for reproducible image generation\n * When the same seed is used with the same prompt and settings,\n * you should get similar (though not identical) results\n */\n seed?: number\n\n /**\n * Whether to add a SynthID watermark to generated images\n * SynthID helps identify AI-generated content\n * @default true\n */\n addWatermark?: boolean\n\n /**\n * Language of the prompt\n * Use ImagePromptLanguage enum values\n */\n language?: ImagePromptLanguage\n\n /**\n * Negative prompt - what to avoid in the generated image\n * Not all models support negative prompts\n */\n negativePrompt?: string\n\n /**\n * Output MIME type for the generated image\n * @default 'image/png'\n */\n outputMimeType?: 'image/png' | 'image/jpeg' | 'image/webp'\n\n /**\n * Compression quality for JPEG outputs (0-100)\n * Higher values mean better quality but larger file sizes\n * @default 75\n */\n outputCompressionQuality?: number\n\n /**\n * Controls how much the model adheres to the text prompt\n * Large values increase output and prompt alignment,\n * but may compromise image quality\n */\n guidanceScale?: number\n\n /**\n * Whether to use the prompt rewriting logic\n */\n enhancePrompt?: boolean\n\n /**\n * Whether to report the safety scores of each generated image\n * and the positive prompt in the response\n */\n includeSafetyAttributes?: boolean\n\n /**\n * Whether to include the Responsible AI filter reason\n * if the image is filtered out of the response\n */\n includeRaiReason?: boolean\n\n /**\n * Cloud Storage URI used to store the generated images\n */\n outputGcsUri?: string\n\n /**\n * User specified labels to track billing usage\n */\n labels?: Record<string, string>\n}\n\n/**\n * Model-specific provider options mapping\n * Currently all Imagen models use the same options structure\n */\nexport type GeminiImageModelProviderOptionsByName = {\n [K in GeminiImageModels]: GeminiImageProviderOptions\n}\n\n/**\n * Supported size strings for Gemini Imagen models\n * These map to aspect ratios internally\n */\nexport type GeminiImageSize =\n | '1024x1024'\n | '512x512'\n | '1024x768'\n | '1536x1024'\n | '1792x1024'\n | '1920x1080'\n | '768x1024'\n | '1024x1536'\n | '1024x1792'\n | '1080x1920'\n\n/**\n * Aspect ratios supported by Gemini native image models (via generateContent API).\n * Matches the SDK's ImageConfig.aspectRatio values.\n */\nexport type GeminiNativeImageAspectRatio =\n | '1:1'\n | '2:3'\n | '3:2'\n | '3:4'\n | '4:3'\n | '9:16'\n | '16:9'\n | '21:9'\n\n/**\n * Resolution tiers for Gemini native image models.\n * Matches the SDK's ImageConfig.imageSize values.\n */\nexport type GeminiNativeImageResolution = '1K' | '2K' | '4K'\n\n/**\n * Template literal size type for Gemini native image models: \"16:9_4K\", \"1:1_2K\", etc.\n */\nexport type GeminiNativeImageSize =\n `${GeminiNativeImageAspectRatio}_${GeminiNativeImageResolution}`\n\n/**\n * Gemini native image models that use the generateContent API path.\n * These models support template literal sizes (aspectRatio_resolution).\n */\nexport type GeminiNativeImageModels =\n | 'gemini-3.1-flash-image-preview'\n | 'gemini-3.1-flash-lite-image'\n | 'gemini-3-pro-image-preview'\n | 'gemini-2.5-flash-image'\n\n/**\n * Model-specific size options mapping.\n * Gemini native image models use template literal sizes, Imagen models use pixel sizes.\n */\nexport type GeminiImageModelSizeByName = {\n [K in GeminiNativeImageModels]: GeminiNativeImageSize\n} & {\n [K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: GeminiImageSize\n}\n\n/**\n * Per-model prompt input modalities. Gemini-native image models accept image\n * parts in the multimodal prompt (image-conditioned generation via\n * generateContent); Imagen models are strictly text-to-image, so their\n * `prompt` is constrained to text at compile time.\n */\nexport type GeminiImageModelInputModalitiesByName = {\n [K in GeminiNativeImageModels]: readonly ['image']\n} & {\n [K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: readonly []\n}\n\n/**\n * Valid sizes for Gemini Imagen models\n * Gemini uses aspect ratios, but we map common WIDTHxHEIGHT formats to aspect ratios\n * These are approximate mappings based on common image dimensions\n */\nexport const GEMINI_SIZE_TO_ASPECT_RATIO: Record<string, GeminiAspectRatio> = {\n // Square\n '1024x1024': '1:1',\n '512x512': '1:1',\n // Landscape\n '1024x768': '4:3',\n '1536x1024': '4:3',\n '1792x1024': '16:9',\n '1920x1080': '16:9',\n // Portrait\n '768x1024': '3:4',\n '1024x1536': '3:4', // Inverted\n '1024x1792': '9:16',\n '1080x1920': '9:16',\n}\n\n/**\n * Maps a WIDTHxHEIGHT size string to a Gemini aspect ratio\n * Returns undefined if the size cannot be mapped\n */\nexport function sizeToAspectRatio(\n size: string | undefined,\n): GeminiAspectRatio | undefined {\n if (!size) return undefined\n return GEMINI_SIZE_TO_ASPECT_RATIO[size]\n}\n\n/**\n * Validates that the provided size can be mapped to an aspect ratio\n * Throws an error if the size is invalid\n */\nexport function validateImageSize(\n model: string,\n size: string | undefined,\n): void {\n if (!size) return\n\n const aspectRatio = sizeToAspectRatio(size)\n if (!aspectRatio) {\n const validSizes = Object.keys(GEMINI_SIZE_TO_ASPECT_RATIO)\n throw new Error(\n `Invalid size \"${size}\" for model \"${model}\". ` +\n `Gemini Imagen uses aspect ratios. Valid sizes that map to aspect ratios: ${validSizes.join(', ')}. ` +\n `Alternatively, use providerOptions.aspectRatio directly with values: 1:1, 3:4, 4:3, 9:16, 16:9, 9:21, 21:9`,\n )\n }\n}\n\n/**\n * Per-model caps on images per request.\n * The Imagen 4 family all support up to 4 images per request via the Gemini\n * API (the rumored 8-image tier is Vertex-only and isn't reachable through\n * @google/genai today). Unknown models fall through to the shared cap\n * defined below.\n *\n * @see https://ai.google.dev/gemini-api/docs/imagen\n */\nconst IMAGEN_MAX_IMAGES_BY_MODEL: Record<string, number> = {\n 'imagen-4.0-generate-001': 4,\n 'imagen-4.0-ultra-generate-001': 4,\n 'imagen-4.0-fast-generate-001': 4,\n}\n\nconst DEFAULT_IMAGEN_MAX_IMAGES = 4\n\n/**\n * Validates the number of images requested against the model's known cap.\n * Uses a per-model table where available and falls back to the shared\n * default otherwise — no more \"some support up to 8\" comments that don't\n * match the error message.\n */\nexport function validateNumberOfImages(\n model: string,\n numberOfImages: number | undefined,\n): void {\n if (numberOfImages === undefined) return\n\n const maxImages =\n IMAGEN_MAX_IMAGES_BY_MODEL[model] ?? DEFAULT_IMAGEN_MAX_IMAGES\n if (numberOfImages < 1 || numberOfImages > maxImages) {\n throw new Error(\n `Invalid numberOfImages \"${numberOfImages}\" for model \"${model}\". ` +\n `Must be between 1 and ${maxImages}.`,\n )\n }\n}\n\n/**\n * Validates the prompt is not empty\n */\nexport function validatePrompt(options: {\n prompt: string\n model: string\n}): void {\n const { prompt, model } = options\n if (!prompt || prompt.trim().length === 0) {\n throw new Error(`Prompt cannot be empty for model \"${model}\".`)\n }\n}\n\n/**\n * Parses a Gemini native image size string into its components.\n * Format: \"aspectRatio_resolution\" e.g. \"16:9_4K\" → { aspectRatio: \"16:9\", resolution: \"4K\" }\n */\nexport function parseNativeImageSize(\n size: string,\n): { aspectRatio: string; resolution: string } | undefined {\n const match = size.match(/^(\\d+:\\d+)_(.+)$/)\n const [, aspectRatio, resolution] = match ?? []\n if (aspectRatio === undefined || resolution === undefined) return undefined\n return { aspectRatio, resolution }\n}\n"],"mappings":";;;;;;AAiNA,IAAa,8BAAiE;CAE5E,aAAa;CACb,WAAW;CAEX,YAAY;CACZ,aAAa;CACb,aAAa;CACb,aAAa;CAEb,YAAY;CACZ,aAAa;CACb,aAAa;CACb,aAAa;AACf;;;;;AAMA,SAAgB,kBACd,MAC+B;CAC/B,IAAI,CAAC,MAAM,OAAO,KAAA;CAClB,OAAO,4BAA4B;AACrC;;;;;AAMA,SAAgB,kBACd,OACA,MACM;CACN,IAAI,CAAC,MAAM;CAGX,IAAI,CADgB,kBAAkB,IACjC,GAAa;EAChB,MAAM,aAAa,OAAO,KAAK,2BAA2B;EAC1D,MAAM,IAAI,MACR,iBAAiB,KAAK,eAAe,MAAM,8EACmC,WAAW,KAAK,IAAI,EAAE,6GAEtG;CACF;AACF;;;;;;;;;;AAWA,IAAM,6BAAqD;CACzD,2BAA2B;CAC3B,iCAAiC;CACjC,gCAAgC;AAClC;AAEA,IAAM,4BAA4B;;;;;;;AAQlC,SAAgB,uBACd,OACA,gBACM;CACN,IAAI,mBAAmB,KAAA,GAAW;CAElC,MAAM,YACJ,2BAA2B,UAAU;CACvC,IAAI,iBAAiB,KAAK,iBAAiB,WACzC,MAAM,IAAI,MACR,2BAA2B,eAAe,eAAe,MAAM,2BACpC,UAAU,EACvC;AAEJ;;;;AAKA,SAAgB,eAAe,SAGtB;CACP,MAAM,EAAE,QAAQ,UAAU;CAC1B,IAAI,CAAC,UAAU,OAAO,KAAK,CAAC,CAAC,WAAW,GACtC,MAAM,IAAI,MAAM,qCAAqC,MAAM,GAAG;AAElE;;;;;AAMA,SAAgB,qBACd,MACyD;CAEzD,MAAM,GAAG,aAAa,cADR,KAAK,MAAM,kBACW,KAAS,CAAC;CAC9C,IAAI,gBAAgB,KAAA,KAAa,eAAe,KAAA,GAAW,OAAO,KAAA;CAClE,OAAO;EAAE;EAAa;CAAW;AACnC"}
|
|
1
|
+
{"version":3,"file":"image-provider-options.js","names":[],"sources":["../../../src/image/image-provider-options.ts"],"sourcesContent":["import type { GeminiImageModels } from '../model-meta'\nimport type {\n ContentUnion,\n ImageConfig,\n ImagePromptLanguage,\n PersonGeneration,\n SafetyFilterLevel,\n SafetySetting,\n ThinkingConfig,\n} from '@google/genai'\n\n// Re-export SDK types so users can use them directly\nexport type {\n ContentUnion,\n ImageConfig,\n ImagePromptLanguage,\n PersonGeneration,\n SafetyFilterLevel,\n SafetySetting,\n ThinkingConfig,\n}\n\n/**\n * Gemini Imagen aspect ratio options\n * Controls the aspect ratio of generated images\n */\nexport type GeminiAspectRatio =\n | '1:1'\n | '3:4'\n | '4:3'\n | '9:16'\n | '16:9'\n | '9:21'\n | '21:9'\n\n/**\n * Provider options for Gemini image generation\n * These options match the @google/genai GenerateImagesConfig interface\n * and can be spread directly into the API request.\n */\nexport interface GeminiImageProviderOptions {\n /**\n * The aspect ratio of generated images\n * @default '1:1'\n */\n aspectRatio?: GeminiAspectRatio\n\n /**\n * Controls whether people can appear in generated images\n * Use PersonGeneration enum values: DONT_ALLOW, ALLOW_ADULT, ALLOW_ALL\n * @default 'ALLOW_ADULT'\n */\n personGeneration?: PersonGeneration\n\n /**\n * Safety filter level for content filtering\n * Use SafetyFilterLevel enum values\n */\n safetyFilterLevel?: SafetyFilterLevel\n\n /**\n * Optional seed for reproducible image generation\n * When the same seed is used with the same prompt and settings,\n * you should get similar (though not identical) results\n */\n seed?: number\n\n /**\n * Whether to add a SynthID watermark to generated images\n * SynthID helps identify AI-generated content\n * @default true\n */\n addWatermark?: boolean\n\n /**\n * Language of the prompt\n * Use ImagePromptLanguage enum values\n */\n language?: ImagePromptLanguage\n\n /**\n * Negative prompt - what to avoid in the generated image\n * Not all models support negative prompts\n */\n negativePrompt?: string\n\n /**\n * Output MIME type for the generated image\n * @default 'image/png'\n */\n outputMimeType?: 'image/png' | 'image/jpeg' | 'image/webp'\n\n /**\n * Compression quality for JPEG outputs (0-100)\n * Higher values mean better quality but larger file sizes\n * @default 75\n */\n outputCompressionQuality?: number\n\n /**\n * Controls how much the model adheres to the text prompt\n * Large values increase output and prompt alignment,\n * but may compromise image quality\n */\n guidanceScale?: number\n\n /**\n * Whether to use the prompt rewriting logic\n */\n enhancePrompt?: boolean\n\n /**\n * Whether to report the safety scores of each generated image\n * and the positive prompt in the response\n */\n includeSafetyAttributes?: boolean\n\n /**\n * Whether to include the Responsible AI filter reason\n * if the image is filtered out of the response\n */\n includeRaiReason?: boolean\n\n /**\n * Cloud Storage URI used to store the generated images\n */\n outputGcsUri?: string\n\n /**\n * User specified labels to track billing usage\n */\n labels?: Record<string, string>\n}\n\n/**\n * Provider options for Gemini native image models (Nano Banana and friends).\n *\n * These models are served by `generateContent`, not `generateImages`, so they\n * are configured by @google/genai's `GenerateContentConfig` — a different\n * shape from the Imagen-only {@link GeminiImageProviderOptions} above. Only\n * the `GenerateContentConfig` fields with clear image-generation semantics are\n * surfaced; sampling knobs (`temperature`, `topK`, …) and chat-only plumbing\n * (`tools`, `responseSchema`, …) are deliberately left out.\n *\n * `responseModalities` is intentionally absent: the adapter always requests\n * `['TEXT', 'IMAGE']`, and letting a caller override it would silently disable\n * image output on an image-generation call.\n */\nexport interface GeminiNativeImageProviderOptions {\n /**\n * Optional seed for reproducible image generation\n * When the same seed is used with the same prompt and settings,\n * you should get similar (though not identical) results\n */\n seed?: number\n\n /**\n * Per-category safety thresholds applied to the request\n * Each entry pairs a HarmCategory with a HarmBlockThreshold\n */\n safetySettings?: Array<SafetySetting>\n\n /**\n * Controls the model's internal reasoning before it emits an image\n * Use to raise or disable the thinking budget on models that support it\n */\n thinkingConfig?: ThinkingConfig\n\n /**\n * Native image output controls. Merged over the values derived from the\n * portable `size` option, so fields set here win per field while the rest\n * of `size` is preserved.\n *\n * Only `aspectRatio` and `imageSize` are accepted on the Gemini Developer\n * API. Other SDK `ImageConfig` keys throw on this surface.\n */\n imageConfig?: GeminiNativeImageConfig\n\n /**\n * System-level instructions that steer the model for the whole request,\n * e.g. a house art direction applied on top of the per-call prompt\n */\n systemInstruction?: ContentUnion\n}\n\n/**\n * Every provider-option field this adapter understands, across both API\n * paths. Used as the adapter's base (model-agnostic) option type; the\n * per-model map below is what narrows a given model to the half that\n * actually applies to it.\n */\nexport type GeminiAnyImageProviderOptions = GeminiImageProviderOptions &\n GeminiNativeImageProviderOptions\n\n/**\n * Model-specific provider options mapping.\n * Gemini native image models go through `generateContent` and take\n * `GenerateContentConfig` fields; Imagen models go through `generateImages`\n * and take `GenerateImagesConfig` fields. Mirrors the native/Imagen split in\n * {@link GeminiImageModelSizeByName} and\n * {@link GeminiImageModelInputModalitiesByName}.\n */\nexport type GeminiImageModelProviderOptionsByName = {\n [K in GeminiNativeImageModels]: GeminiNativeImageProviderOptions\n} & {\n [K in Exclude<\n GeminiImageModels,\n GeminiNativeImageModels\n >]: GeminiImageProviderOptions\n}\n\n/**\n * Supported size strings for Gemini Imagen models\n * These map to aspect ratios internally\n */\nexport type GeminiImageSize =\n | '1024x1024'\n | '512x512'\n | '1024x768'\n | '1536x1024'\n | '1792x1024'\n | '1920x1080'\n | '768x1024'\n | '1024x1536'\n | '1024x1792'\n | '1080x1920'\n\n/**\n * The ten aspect ratios every Gemini native image model accepts.\n *\n * Note `9:21` is deliberately absent: it exists only on Vertex / Cloud and is\n * rejected by the Gemini API (`generateContent`), which is the surface this\n * adapter targets.\n *\n * @see https://ai.google.dev/gemini-api/docs/image-generation\n */\nexport type GeminiStandardImageAspectRatio =\n | '1:1'\n | '2:3'\n | '3:2'\n | '3:4'\n | '4:3'\n | '4:5'\n | '5:4'\n | '9:16'\n | '16:9'\n | '21:9'\n\n/**\n * The ten standard ratios plus the four extreme banner/strip ratios that only\n * the Gemini 3.1 Flash Image models accept — 14 values, matching the\n * `generateContent` `ImageConfig.aspectRatio` field union.\n *\n * @see https://ai.google.dev/api/generate-content\n */\nexport type GeminiExtendedImageAspectRatio =\n | GeminiStandardImageAspectRatio\n | '1:4'\n | '4:1'\n | '1:8'\n | '8:1'\n\n/**\n * Sizes for `gemini-3.1-flash-image` (and its shut-down `-preview` alias):\n * all 14 aspect ratios at 512 / 1K / 2K / 4K. `512` is the wire token for the\n * 0.5K tier — not `512px`, and the `K` is case-sensitive (`1k` is rejected).\n */\nexport type Gemini31FlashImageSize =\n `${GeminiExtendedImageAspectRatio}_${'512' | '1K' | '2K' | '4K'}`\n\n/**\n * Sizes for `gemini-3.1-flash-lite-image`: all 14 aspect ratios, 1K only.\n * 2K and 4K are unsupported on this model.\n *\n * The four banner ratios (`1:4`, `4:1`, `1:8`, `8:1`) come from the Cloud\n * model page. The Gemini API page states a count of 14 but does not list them.\n *\n * @see https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/gemini/3-1-flash-lite-image\n * @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image\n */\nexport type Gemini31FlashLiteImageSize = `${GeminiExtendedImageAspectRatio}_1K`\n\n/**\n * Sizes for `gemini-3-pro-image` (and its shut-down `-preview` alias): the ten\n * standard aspect ratios at 1K / 2K / 4K. Pro has no 512 tier and none of the\n * extreme banner ratios on the Gemini API.\n */\nexport type Gemini3ProImageSize =\n `${GeminiStandardImageAspectRatio}_${'1K' | '2K' | '4K'}`\n\n/**\n * Sizes for `gemini-2.5-flash-image`: a bare aspect ratio with no resolution\n * suffix, e.g. `'16:9'`. Google documents no `image_size` value or default for\n * this model — it emits a single fixed 1024px-class output — so the adapter\n * sends `imageConfig.aspectRatio` and omits `imageSize` entirely rather than\n * guessing a tier the API never documented.\n */\nexport type Gemini25FlashImageSize = GeminiStandardImageAspectRatio\n\n/**\n * `imageConfig` fields the Gemini Developer API accepts on `generateContent`.\n * Other `@google/genai` `ImageConfig` keys (`personGeneration`,\n * `outputMimeType`, and more) throw on this surface.\n */\nexport type GeminiNativeImageConfig = {\n aspectRatio?: GeminiExtendedImageAspectRatio\n imageSize?: '512' | '1K' | '2K' | '4K'\n}\n\n/**\n * Any size accepted by any Gemini native image model. Prefer the per-model\n * narrowing in {@link GeminiImageModelSizeByName} — this union is the widest\n * possible set and accepts combinations no single model supports.\n */\nexport type GeminiNativeImageSize =\n | Gemini31FlashImageSize\n | Gemini31FlashLiteImageSize\n | Gemini3ProImageSize\n | Gemini25FlashImageSize\n\n/**\n * Gemini native image models that use the generateContent API path.\n * These models take an aspect-ratio-based size rather than Imagen's\n * WIDTHxHEIGHT pixel strings.\n *\n * This array is the single source of truth for the native/Imagen split: the\n * `GeminiNativeImageModels` union and the per-model option/size/modality maps\n * all derive from it. The `satisfies` clause makes a typo (or a name that\n * is not a known image model) a build error rather than a phantom key on every\n * per-model map.\n *\n * It is also the single source of truth for the adapter's runtime routing\n * — see {@link isGeminiNativeImageModel}. Adding a new `gemini-*` image model\n * means adding it here as well as to `GEMINI_IMAGE_MODELS` in model-meta.\n * Until it is listed here it routes to the Imagen API instead and fails\n * loudly on the first call, rather than silently taking the wrong option\n * shape.\n */\nexport const GEMINI_NATIVE_IMAGE_MODELS = [\n 'gemini-3.1-flash-image',\n 'gemini-3.1-flash-image-preview',\n 'gemini-3.1-flash-lite-image',\n 'gemini-3-pro-image',\n 'gemini-3-pro-image-preview',\n 'gemini-2.5-flash-image',\n] as const satisfies ReadonlyArray<GeminiImageModels>\n\nexport type GeminiNativeImageModels =\n (typeof GEMINI_NATIVE_IMAGE_MODELS)[number]\n\nconst NATIVE_IMAGE_MODEL_NAMES: ReadonlySet<string> = new Set(\n GEMINI_NATIVE_IMAGE_MODELS,\n)\n\n/**\n * Runtime counterpart to {@link GeminiNativeImageModels} — decides which of\n * the two Gemini image APIs a model goes to.\n *\n * Membership in {@link GEMINI_NATIVE_IMAGE_MODELS}, not a `gemini-` prefix\n * test, so the runtime route and the type-level split cannot drift apart. An\n * id this package does not know about reaches the Imagen endpoint and fails\n * there, which is the intended signal to add the model here rather than to\n * have it silently take the native path with Imagen-shaped option types.\n */\nexport function isGeminiNativeImageModel(model: string): boolean {\n return NATIVE_IMAGE_MODEL_NAMES.has(model)\n}\n\n/**\n * Model-specific size options mapping. Each native model gets its own ratio ×\n * resolution set (they genuinely differ); Imagen models use pixel sizes.\n */\nexport type GeminiImageModelSizeByName = {\n 'gemini-3.1-flash-image': Gemini31FlashImageSize\n 'gemini-3.1-flash-image-preview': Gemini31FlashImageSize\n 'gemini-3.1-flash-lite-image': Gemini31FlashLiteImageSize\n 'gemini-3-pro-image': Gemini3ProImageSize\n 'gemini-3-pro-image-preview': Gemini3ProImageSize\n 'gemini-2.5-flash-image': Gemini25FlashImageSize\n} & {\n [K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: GeminiImageSize\n}\n\n/**\n * Per-model prompt input modalities. Gemini-native image models accept image\n * parts in the multimodal prompt (image-conditioned generation via\n * generateContent); Imagen models are strictly text-to-image, so their\n * `prompt` is constrained to text at compile time.\n */\nexport type GeminiImageModelInputModalitiesByName = {\n [K in GeminiNativeImageModels]: readonly ['image']\n} & {\n [K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: readonly []\n}\n\n/**\n * Valid sizes for Gemini Imagen models\n * Gemini uses aspect ratios, but we map common WIDTHxHEIGHT formats to aspect ratios\n * These are approximate mappings based on common image dimensions\n */\nexport const GEMINI_SIZE_TO_ASPECT_RATIO: Record<string, GeminiAspectRatio> = {\n // Square\n '1024x1024': '1:1',\n '512x512': '1:1',\n // Landscape\n '1024x768': '4:3',\n '1536x1024': '4:3',\n '1792x1024': '16:9',\n '1920x1080': '16:9',\n // Portrait\n '768x1024': '3:4',\n '1024x1536': '3:4', // Inverted\n '1024x1792': '9:16',\n '1080x1920': '9:16',\n}\n\n/**\n * Maps a WIDTHxHEIGHT size string to a Gemini aspect ratio\n * Returns undefined if the size cannot be mapped\n */\nexport function sizeToAspectRatio(\n size: string | undefined,\n): GeminiAspectRatio | undefined {\n if (!size) return undefined\n return GEMINI_SIZE_TO_ASPECT_RATIO[size]\n}\n\n/**\n * Validates that the provided size can be mapped to an aspect ratio\n * Throws an error if the size is invalid\n */\nexport function validateImageSize(\n model: string,\n size: string | undefined,\n): void {\n if (!size) return\n\n const aspectRatio = sizeToAspectRatio(size)\n if (!aspectRatio) {\n const validSizes = Object.keys(GEMINI_SIZE_TO_ASPECT_RATIO)\n throw new Error(\n `Invalid size \"${size}\" for model \"${model}\". ` +\n `Gemini Imagen uses aspect ratios. Valid sizes that map to aspect ratios: ${validSizes.join(', ')}. ` +\n `Alternatively, use providerOptions.aspectRatio directly with values: 1:1, 3:4, 4:3, 9:16, 16:9, 9:21, 21:9`,\n )\n }\n}\n\n/**\n * Per-model caps on images per request.\n * The Imagen 4 family all support up to 4 images per request via the Gemini\n * API (the rumored 8-image tier is Vertex-only and isn't reachable through\n * @google/genai today). Unknown models fall through to the shared cap\n * defined below.\n *\n * @see https://ai.google.dev/gemini-api/docs/imagen\n */\nconst IMAGEN_MAX_IMAGES_BY_MODEL: Record<string, number> = {\n 'imagen-4.0-generate-001': 4,\n 'imagen-4.0-ultra-generate-001': 4,\n 'imagen-4.0-fast-generate-001': 4,\n}\n\nconst DEFAULT_IMAGEN_MAX_IMAGES = 4\n\n/**\n * Validates the number of images requested against the model's known cap.\n * Uses a per-model table where available and falls back to the shared\n * default otherwise — no more \"some support up to 8\" comments that don't\n * match the error message.\n */\nexport function validateNumberOfImages(\n model: string,\n numberOfImages: number | undefined,\n): void {\n if (numberOfImages === undefined) return\n\n const maxImages =\n IMAGEN_MAX_IMAGES_BY_MODEL[model] ?? DEFAULT_IMAGEN_MAX_IMAGES\n if (numberOfImages < 1 || numberOfImages > maxImages) {\n throw new Error(\n `Invalid numberOfImages \"${numberOfImages}\" for model \"${model}\". ` +\n `Must be between 1 and ${maxImages}.`,\n )\n }\n}\n\n/**\n * Validates the prompt is not empty\n */\nexport function validatePrompt(options: {\n prompt: string\n model: string\n}): void {\n const { prompt, model } = options\n if (!prompt || prompt.trim().length === 0) {\n throw new Error(`Prompt cannot be empty for model \"${model}\".`)\n }\n}\n\n/**\n * Parses a Gemini native image size string into its components.\n *\n * Format: `\"aspectRatio_resolution\"`, e.g. `\"16:9_4K\"` →\n * `{ aspectRatio: \"16:9\", resolution: \"4K\" }`.\n *\n * The resolution suffix is optional: `gemini-2.5-flash-image` takes a bare\n * aspect ratio (`\"16:9\"` → `{ aspectRatio: \"16:9\" }`) because Google documents\n * no `image_size` for it, and the caller must then omit `imageSize` from the\n * request rather than substituting a default.\n */\nexport function parseNativeImageSize(\n size: string,\n): { aspectRatio: string; resolution?: string } | undefined {\n const match = size.match(/^(\\d+:\\d+)(?:_(.+))?$/)\n const [, aspectRatio, resolution] = match ?? []\n if (aspectRatio === undefined) return undefined\n return {\n aspectRatio,\n ...(resolution !== undefined && { resolution }),\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;AAkVA,IAAa,6BAA6B;CACxC;CACA;CACA;CACA;CACA;CACA;AACF;AAKA,IAAM,2BAAgD,IAAI,IACxD,0BACF;;;;;;;;;;;AAYA,SAAgB,yBAAyB,OAAwB;CAC/D,OAAO,yBAAyB,IAAI,KAAK;AAC3C;;;;;;AAkCA,IAAa,8BAAiE;CAE5E,aAAa;CACb,WAAW;CAEX,YAAY;CACZ,aAAa;CACb,aAAa;CACb,aAAa;CAEb,YAAY;CACZ,aAAa;CACb,aAAa;CACb,aAAa;AACf;;;;;AAMA,SAAgB,kBACd,MAC+B;CAC/B,IAAI,CAAC,MAAM,OAAO,KAAA;CAClB,OAAO,4BAA4B;AACrC;;;;;AAMA,SAAgB,kBACd,OACA,MACM;CACN,IAAI,CAAC,MAAM;CAGX,IAAI,CADgB,kBAAkB,IACjC,GAAa;EAChB,MAAM,aAAa,OAAO,KAAK,2BAA2B;EAC1D,MAAM,IAAI,MACR,iBAAiB,KAAK,eAAe,MAAM,8EACmC,WAAW,KAAK,IAAI,EAAE,6GAEtG;CACF;AACF;;;;;;;;;;AAWA,IAAM,6BAAqD;CACzD,2BAA2B;CAC3B,iCAAiC;CACjC,gCAAgC;AAClC;AAEA,IAAM,4BAA4B;;;;;;;AAQlC,SAAgB,uBACd,OACA,gBACM;CACN,IAAI,mBAAmB,KAAA,GAAW;CAElC,MAAM,YACJ,2BAA2B,UAAU;CACvC,IAAI,iBAAiB,KAAK,iBAAiB,WACzC,MAAM,IAAI,MACR,2BAA2B,eAAe,eAAe,MAAM,2BACpC,UAAU,EACvC;AAEJ;;;;AAKA,SAAgB,eAAe,SAGtB;CACP,MAAM,EAAE,QAAQ,UAAU;CAC1B,IAAI,CAAC,UAAU,OAAO,KAAK,CAAC,CAAC,WAAW,GACtC,MAAM,IAAI,MAAM,qCAAqC,MAAM,GAAG;AAElE;;;;;;;;;;;;AAaA,SAAgB,qBACd,MAC0D;CAE1D,MAAM,GAAG,aAAa,cADR,KAAK,MAAM,uBACW,KAAS,CAAC;CAC9C,IAAI,gBAAgB,KAAA,GAAW,OAAO,KAAA;CACtC,OAAO;EACL;EACA,GAAI,eAAe,KAAA,KAAa,EAAE,WAAW;CAC/C;AACF"}
|
package/dist/esm/index.d.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
export { GeminiTextAdapter, createGeminiChat, geminiText, type GeminiTextConfig, type GeminiTextProviderOptions, } from './adapters/text.js';
|
|
2
2
|
export { createGeminiSummarize, geminiSummarize, type GeminiSummarizeConfig, type GeminiSummarizeModel, } from './adapters/summarize.js';
|
|
3
3
|
export { GeminiImageAdapter, createGeminiImage, geminiImage, type GeminiImageConfig, } from './adapters/image.js';
|
|
4
|
-
export type { GeminiImageProviderOptions, GeminiImageModelProviderOptionsByName, GeminiAspectRatio, PersonGeneration, SafetyFilterLevel, ImagePromptLanguage, } from './image/image-provider-options.js';
|
|
4
|
+
export type { GeminiImageProviderOptions, GeminiNativeImageConfig, GeminiNativeImageProviderOptions, GeminiAnyImageProviderOptions, GeminiImageModelProviderOptionsByName, GeminiAspectRatio, GeminiImageModelSizeByName, GeminiStandardImageAspectRatio, GeminiExtendedImageAspectRatio, Gemini31FlashImageSize, Gemini31FlashLiteImageSize, Gemini3ProImageSize, Gemini25FlashImageSize, GeminiNativeImageSize, PersonGeneration, SafetyFilterLevel, ImagePromptLanguage, SafetySetting, ThinkingConfig, ImageConfig, ContentUnion, } from './image/image-provider-options.js';
|
|
5
|
+
export { HarmBlockThreshold, HarmCategory } from '@google/genai';
|
|
5
6
|
export { GeminiEmbeddingAdapter, createGeminiEmbedding, geminiEmbedding, type GeminiEmbeddingConfig, } from './adapters/embedding.js';
|
|
6
7
|
export type { GeminiEmbeddingProviderOptions } from './embedding/embedding-provider-options.js';
|
|
7
8
|
/**
|
|
@@ -21,6 +22,7 @@ export type { GeminiInteractionsVideoModel, GeminiOmniVideoProviderOptions, Gemi
|
|
|
21
22
|
export { GEMINI_MODELS, GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS, } from './model-meta.js';
|
|
22
23
|
export { GEMINI_MODELS as GeminiTextModels } from './model-meta.js';
|
|
23
24
|
export { GEMINI_IMAGE_MODELS as GeminiImageModels } from './model-meta.js';
|
|
25
|
+
export { GEMINI_NATIVE_IMAGE_MODELS, isGeminiNativeImageModel, } from './image/image-provider-options.js';
|
|
24
26
|
export { GEMINI_TTS_MODELS as GeminiTTSModels } from './model-meta.js';
|
|
25
27
|
export { GEMINI_TTS_VOICES as GeminiTTSVoices } from './model-meta.js';
|
|
26
28
|
export { GEMINI_AUDIO_MODELS as GeminiAudioModels } from './model-meta.js';
|
package/dist/esm/index.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { GEMINI_AUDIO_MODELS, GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS, GEMINI_EMBEDDING_MODELS, GEMINI_IMAGE_MODELS, GEMINI_INTERACTIONS_VIDEO_MODELS, GEMINI_MODELS, GEMINI_TTS_MODELS, GEMINI_TTS_VOICES, GEMINI_VIDEO_MODELS } from "./model-meta.js";
|
|
2
2
|
import { GeminiTextAdapter, createGeminiChat, geminiText } from "./adapters/text.js";
|
|
3
3
|
import { createGeminiSummarize, geminiSummarize } from "./adapters/summarize.js";
|
|
4
|
+
import { GEMINI_NATIVE_IMAGE_MODELS, isGeminiNativeImageModel } from "./image/image-provider-options.js";
|
|
4
5
|
import { GeminiImageAdapter, createGeminiImage, geminiImage } from "./adapters/image.js";
|
|
5
6
|
import { GeminiEmbeddingAdapter, createGeminiEmbedding, geminiEmbedding } from "./adapters/embedding.js";
|
|
6
7
|
import { GeminiTTSAdapter, createGeminiSpeech, geminiSpeech } from "./adapters/tts.js";
|
|
@@ -10,4 +11,5 @@ import { GeminiVideoAdapter, createGeminiVideo, geminiVideo } from "./adapters/v
|
|
|
10
11
|
import { geminiRealtimeToken } from "./realtime/token.js";
|
|
11
12
|
import { geminiRealtime } from "./realtime/adapter.js";
|
|
12
13
|
import "./realtime/index.js";
|
|
13
|
-
|
|
14
|
+
import { HarmBlockThreshold, HarmCategory } from "@google/genai";
|
|
15
|
+
export { GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS, GEMINI_EMBEDDING_MODELS, GEMINI_MODELS, GEMINI_NATIVE_IMAGE_MODELS, GEMINI_VIDEO_DURATIONS, GeminiAudioAdapter, GEMINI_AUDIO_MODELS as GeminiAudioModels, GeminiEmbeddingAdapter, GeminiImageAdapter, GEMINI_IMAGE_MODELS as GeminiImageModels, GEMINI_INTERACTIONS_VIDEO_MODELS as GeminiInteractionsVideoModels, GeminiTTSAdapter, GEMINI_TTS_MODELS as GeminiTTSModels, GEMINI_TTS_VOICES as GeminiTTSVoices, GeminiTextAdapter, GEMINI_MODELS as GeminiTextModels, GeminiVideoAdapter, GEMINI_VIDEO_MODELS as GeminiVideoModels, HarmBlockThreshold, HarmCategory, createGeminiAudio, createGeminiChat, createGeminiEmbedding, createGeminiImage, createGeminiSpeech, createGeminiSummarize, createGeminiVideo, geminiAudio, geminiEmbedding, geminiImage, geminiRealtime, geminiRealtimeToken, geminiSpeech, geminiSummarize, geminiText, geminiVideo, getGeminiVideoDurationOptions, isGeminiNativeImageModel, isInteractionsVideoModel };
|
package/dist/esm/model-meta.d.ts
CHANGED
|
@@ -233,8 +233,21 @@ export declare const GEMINI_MODELS: readonly ["gemini-3.7-flash", "gemini-3.6-fl
|
|
|
233
233
|
*/
|
|
234
234
|
export declare const GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS: Set<string>;
|
|
235
235
|
export type GeminiModels = (typeof GEMINI_MODELS)[number];
|
|
236
|
-
|
|
237
|
-
|
|
236
|
+
/**
|
|
237
|
+
* @deprecated Shut down 2026-06-25. Use `gemini-3.1-flash-image`.
|
|
238
|
+
*/
|
|
239
|
+
type Gemini31FlashImagePreviewModel = 'gemini-3.1-flash-image-preview';
|
|
240
|
+
/**
|
|
241
|
+
* @deprecated Shut down 2026-06-25. Use `gemini-3-pro-image`.
|
|
242
|
+
*/
|
|
243
|
+
type Gemini3ProImagePreviewModel = 'gemini-3-pro-image-preview';
|
|
244
|
+
export type GeminiImageModels = Exclude<(typeof GEMINI_IMAGE_MODELS)[number], Gemini31FlashImagePreviewModel | Gemini3ProImagePreviewModel> | Gemini31FlashImagePreviewModel | Gemini3ProImagePreviewModel;
|
|
245
|
+
/**
|
|
246
|
+
* Image generation models. GA ids come first; the trailing `-preview` ids are
|
|
247
|
+
* shut-down aliases kept only so existing code keeps compiling — new code
|
|
248
|
+
* should use the GA id above its alias.
|
|
249
|
+
*/
|
|
250
|
+
export declare const GEMINI_IMAGE_MODELS: readonly ["gemini-3.1-flash-image", "gemini-3.1-flash-lite-image", "gemini-3-pro-image", "gemini-2.5-flash-image", "imagen-4.0-generate-001", "imagen-4.0-fast-generate-001", "imagen-4.0-ultra-generate-001", "gemini-3.1-flash-image-preview", "gemini-3-pro-image-preview"];
|
|
238
251
|
/**
|
|
239
252
|
* Text-to-speech models
|
|
240
253
|
* @experimental Gemini TTS is an experimental feature and may change.
|
package/dist/esm/model-meta.js
CHANGED
|
@@ -65,7 +65,38 @@ var GEMINI_3_FLASH = {
|
|
|
65
65
|
output: { normal: 3 }
|
|
66
66
|
}
|
|
67
67
|
};
|
|
68
|
+
/**
|
|
69
|
+
* Gemini 3 Pro Image ("Nano Banana Pro") — GA. Accepts the ten standard
|
|
70
|
+
* aspect ratios at 1K / 2K / 4K.
|
|
71
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3-pro-image
|
|
72
|
+
*/
|
|
68
73
|
var GEMINI_3_PRO_IMAGE = {
|
|
74
|
+
name: "gemini-3-pro-image",
|
|
75
|
+
max_input_tokens: 65536,
|
|
76
|
+
max_output_tokens: 32768,
|
|
77
|
+
knowledge_cutoff: "2025-01-01",
|
|
78
|
+
supports: {
|
|
79
|
+
input: ["text", "image"],
|
|
80
|
+
output: ["text", "image"],
|
|
81
|
+
capabilities: [
|
|
82
|
+
"batch_api",
|
|
83
|
+
"structured_output",
|
|
84
|
+
"thinking"
|
|
85
|
+
],
|
|
86
|
+
tools: ["google_search"]
|
|
87
|
+
},
|
|
88
|
+
pricing: {
|
|
89
|
+
input: { normal: 2 },
|
|
90
|
+
output: { normal: .134 }
|
|
91
|
+
}
|
|
92
|
+
};
|
|
93
|
+
/**
|
|
94
|
+
* @deprecated `gemini-3-pro-image-preview` was shut down on 2026-06-25. Use
|
|
95
|
+
* the GA id `gemini-3-pro-image` instead — the preview id now 404s.
|
|
96
|
+
* Kept in the model union so existing code still compiles.
|
|
97
|
+
* @see https://ai.google.dev/gemini-api/docs/deprecations
|
|
98
|
+
*/
|
|
99
|
+
var GEMINI_3_PRO_IMAGE_PREVIEW = {
|
|
69
100
|
name: "gemini-3-pro-image-preview",
|
|
70
101
|
max_input_tokens: 65536,
|
|
71
102
|
max_output_tokens: 32768,
|
|
@@ -85,7 +116,35 @@ var GEMINI_3_PRO_IMAGE = {
|
|
|
85
116
|
output: { normal: .134 }
|
|
86
117
|
}
|
|
87
118
|
};
|
|
119
|
+
/**
|
|
120
|
+
* Gemini 3.1 Flash Image ("Nano Banana 2") — GA. The only native image model
|
|
121
|
+
* that accepts the four extreme banner ratios (1:4, 4:1, 1:8, 8:1) and the
|
|
122
|
+
* 512 (0.5K) resolution tier.
|
|
123
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-image
|
|
124
|
+
*/
|
|
88
125
|
var GEMINI_3_1_FLASH_IMAGE = {
|
|
126
|
+
name: "gemini-3.1-flash-image",
|
|
127
|
+
max_input_tokens: 131072,
|
|
128
|
+
max_output_tokens: 32768,
|
|
129
|
+
knowledge_cutoff: "2025-01-01",
|
|
130
|
+
supports: {
|
|
131
|
+
input: ["text", "image"],
|
|
132
|
+
output: ["text", "image"],
|
|
133
|
+
capabilities: ["batch_api", "thinking"],
|
|
134
|
+
tools: ["google_search"]
|
|
135
|
+
},
|
|
136
|
+
pricing: {
|
|
137
|
+
input: { normal: .5 },
|
|
138
|
+
output: { normal: 3 }
|
|
139
|
+
}
|
|
140
|
+
};
|
|
141
|
+
/**
|
|
142
|
+
* @deprecated `gemini-3.1-flash-image-preview` was shut down on 2026-06-25.
|
|
143
|
+
* Use the GA id `gemini-3.1-flash-image` instead — the preview id now 404s.
|
|
144
|
+
* Kept in the model union so existing code still compiles.
|
|
145
|
+
* @see https://ai.google.dev/gemini-api/docs/deprecations
|
|
146
|
+
*/
|
|
147
|
+
var GEMINI_3_1_FLASH_IMAGE_PREVIEW = {
|
|
89
148
|
name: "gemini-3.1-flash-image-preview",
|
|
90
149
|
max_input_tokens: 65536,
|
|
91
150
|
max_output_tokens: 65536,
|
|
@@ -105,6 +164,10 @@ var GEMINI_3_1_FLASH_IMAGE = {
|
|
|
105
164
|
output: { normal: 1.5 }
|
|
106
165
|
}
|
|
107
166
|
};
|
|
167
|
+
/**
|
|
168
|
+
* Gemini 3.1 Flash Lite Image ("Nano Banana 2 Lite") — GA. 1K output only.
|
|
169
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image
|
|
170
|
+
*/
|
|
108
171
|
var GEMINI_3_1_FLASH_LITE_IMAGE = {
|
|
109
172
|
name: "gemini-3.1-flash-lite-image",
|
|
110
173
|
max_input_tokens: 65536,
|
|
@@ -273,6 +336,17 @@ var GEMINI_2_5_FLASH = {
|
|
|
273
336
|
output: { normal: 2.5 }
|
|
274
337
|
}
|
|
275
338
|
};
|
|
339
|
+
/**
|
|
340
|
+
* Gemini 2.5 Flash Image ("Nano Banana") — still GA, but documented as the
|
|
341
|
+
* legacy member of the family. Google publishes no `image_size` value for it,
|
|
342
|
+
* so its size type is a bare aspect ratio and the adapter sends no
|
|
343
|
+
* `imageConfig.imageSize`.
|
|
344
|
+
* @deprecated `gemini-2.5-flash-image` shuts down on 2026-10-02. Migrate to
|
|
345
|
+
* `gemini-3.1-flash-lite-image` (cheapest successor) or
|
|
346
|
+
* `gemini-3.1-flash-image`. Google's deprecations table still names the
|
|
347
|
+
* already-dead `gemini-3.1-flash-image-preview` as the replacement.
|
|
348
|
+
* @see https://ai.google.dev/gemini-api/docs/deprecations
|
|
349
|
+
*/
|
|
276
350
|
var GEMINI_2_5_FLASH_IMAGE = {
|
|
277
351
|
name: "gemini-2.5-flash-image",
|
|
278
352
|
max_input_tokens: 1048576,
|
|
@@ -720,6 +794,11 @@ var GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS = /* @__PURE__ */ new Set([
|
|
|
720
794
|
GEMINI_3_1_FLASH_LITE.name,
|
|
721
795
|
GEMINI_3_1_FLASH_LITE_PREVIEW.name
|
|
722
796
|
]);
|
|
797
|
+
/**
|
|
798
|
+
* Image generation models. GA ids come first; the trailing `-preview` ids are
|
|
799
|
+
* shut-down aliases kept only so existing code keeps compiling — new code
|
|
800
|
+
* should use the GA id above its alias.
|
|
801
|
+
*/
|
|
723
802
|
var GEMINI_IMAGE_MODELS = [
|
|
724
803
|
GEMINI_3_1_FLASH_IMAGE.name,
|
|
725
804
|
GEMINI_3_1_FLASH_LITE_IMAGE.name,
|
|
@@ -727,7 +806,9 @@ var GEMINI_IMAGE_MODELS = [
|
|
|
727
806
|
GEMINI_2_5_FLASH_IMAGE.name,
|
|
728
807
|
IMAGEN_4_GENERATE.name,
|
|
729
808
|
IMAGEN_4_GENERATE_FAST.name,
|
|
730
|
-
IMAGEN_4_GENERATE_ULTRA.name
|
|
809
|
+
IMAGEN_4_GENERATE_ULTRA.name,
|
|
810
|
+
GEMINI_3_1_FLASH_IMAGE_PREVIEW.name,
|
|
811
|
+
GEMINI_3_PRO_IMAGE_PREVIEW.name
|
|
731
812
|
];
|
|
732
813
|
/**
|
|
733
814
|
* Text-to-speech models
|