@tanstack/ai-gemini 0.23.0 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/adapters/image.d.ts +14 -7
- package/dist/esm/adapters/image.js +27 -56
- package/dist/esm/adapters/image.js.map +1 -1
- package/dist/esm/adapters/text.js.map +1 -1
- package/dist/esm/adapters/video.js +2 -1
- package/dist/esm/adapters/video.js.map +1 -1
- package/dist/esm/experimental/text-interactions/adapter.js +3 -9
- package/dist/esm/experimental/text-interactions/adapter.js.map +1 -1
- package/dist/esm/image/image-provider-options.d.ts +167 -20
- package/dist/esm/image/image-provider-options.js +52 -5
- package/dist/esm/image/image-provider-options.js.map +1 -1
- package/dist/esm/index.d.ts +3 -1
- package/dist/esm/index.js +3 -1
- package/dist/esm/model-meta.d.ts +15 -2
- package/dist/esm/model-meta.js +82 -1
- package/dist/esm/model-meta.js.map +1 -1
- package/dist/esm/realtime/adapter.js +1 -5
- package/dist/esm/realtime/adapter.js.map +1 -1
- package/dist/esm/realtime/client.js.map +1 -1
- package/dist/esm/realtime/token.js +3 -2
- package/dist/esm/realtime/token.js.map +1 -1
- package/dist/esm/realtime/utils.js +4 -1
- package/dist/esm/realtime/utils.js.map +1 -1
- package/dist/esm/tools/tool-converter.js +0 -1
- package/dist/esm/tools/tool-converter.js.map +1 -1
- package/dist/esm/usage.js +1 -3
- package/dist/esm/usage.js.map +1 -1
- package/package.json +5 -5
- package/src/adapters/image.ts +123 -36
- package/src/image/image-provider-options.ts +226 -25
- package/src/index.ts +28 -0
- package/src/model-meta.ts +118 -1
package/dist/esm/usage.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"usage.js","names":[],"sources":["../../src/usage.ts"],"sourcesContent":["import { buildBaseUsage } from '@tanstack/ai'\nimport type { TokenUsage } from '@tanstack/ai'\nimport type {\n GenerateContentResponseUsageMetadata,\n ModalityTokenCount,\n} from '@google/genai'\n\n/**\n * Flattened modality token counts for normalized usage reporting.\n * Maps Gemini's ModalityTokenCount array to individual fields.\n */\nexport interface FlattenedModalityTokens {\n /** Text tokens */\n textTokens?: number\n /** Image tokens */\n imageTokens?: number\n /** Audio tokens */\n audioTokens?: number\n /** Video tokens */\n videoTokens?: number\n /** Document tokens (e.g. PDF inputs) */\n documentTokens?: number\n}\n\n/**\n * Flattens Gemini's ModalityTokenCount array into individual token fields.\n * Extracts TEXT, IMAGE, AUDIO, VIDEO, DOCUMENT modality counts into a\n * normalized structure.\n */\nexport function flattenModalityTokenCounts(\n modalities?: Array<ModalityTokenCount>,\n): FlattenedModalityTokens {\n if (!modalities || modalities.length === 0) {\n return {}\n }\n\n const result: FlattenedModalityTokens = {}\n\n for (const item of modalities) {\n if (!item.modality || item.tokenCount === undefined) {\n continue\n }\n\n const modality = item.modality.toUpperCase()\n const count = item.tokenCount\n\n switch (modality) {\n case 'TEXT':\n result.textTokens = (result.textTokens ?? 0) + count\n break\n case 'IMAGE':\n result.imageTokens = (result.imageTokens ?? 0) + count\n break\n case 'AUDIO':\n result.audioTokens = (result.audioTokens ?? 0) + count\n break\n case 'VIDEO':\n result.videoTokens = (result.videoTokens ?? 0) + count\n break\n case 'DOCUMENT':\n result.documentTokens = (result.documentTokens ?? 0) + count\n break\n }\n }\n\n return result\n}\n\n/**\n * Checks if a FlattenedModalityTokens object has any values set.\n */\nexport function hasModalityTokens(tokens: FlattenedModalityTokens): boolean {\n return (\n tokens.textTokens !== undefined ||\n tokens.imageTokens !== undefined ||\n tokens.audioTokens !== undefined ||\n tokens.videoTokens !== undefined ||\n tokens.documentTokens !== undefined\n )\n}\n\n/**\n * Gemini-specific provider usage details.\n * These fields are unique to Gemini and placed in providerUsageDetails.\n */\nexport type GeminiProviderUsageDetails = {\n /**\n * The traffic type for this request.\n * Can indicate whether request was handled by different service tiers.\n */\n trafficType?: string\n /**\n * Number of tokens in the results from tool executions,\n * which are provided back to the model as input.\n */\n toolUsePromptTokenCount?: number\n /**\n * Detailed breakdown by modality of the token counts from\n * the results of tool executions.\n */\n toolUsePromptTokensDetails?: Array<{\n modality: string\n tokenCount: number\n }>\n /**\n * Detailed breakdown of cache tokens by modality.\n * More granular than the normalized cachedTokens field.\n */\n cacheTokensDetails?: Array<{\n modality: string\n tokenCount: number\n }>\n}\n\n/**\n * Build normalized TokenUsage from Gemini's usageMetadata.\n * Handles modality breakdowns and thinking tokens. Returns `undefined` when the\n * provider reported no usage metadata, so callers omit the field rather than\n * fabricating zeroed totals.\n */\nexport function buildGeminiUsage(\n usageMetadata: GenerateContentResponseUsageMetadata | undefined | null,\n): TokenUsage<GeminiProviderUsageDetails> | undefined {\n if (!usageMetadata) return undefined\n\n const promptTokens = usageMetadata.promptTokenCount ?? 0\n const completionTokens = usageMetadata.candidatesTokenCount ?? 0\n\n const result = buildBaseUsage<GeminiProviderUsageDetails>({\n promptTokens: promptTokens,\n completionTokens: completionTokens,\n totalTokens:\n usageMetadata.totalTokenCount ?? promptTokens + completionTokens,\n })\n\n // Add prompt token details\n // Flatten modality breakdown for prompt\n const promptModalities = flattenModalityTokenCounts(\n usageMetadata.promptTokensDetails,\n )\n const cachedTokens = usageMetadata.cachedContentTokenCount\n\n const promptTokensDetails = {\n ...(hasModalityTokens(promptModalities) ? promptModalities : {}),\n ...(cachedTokens !== undefined && cachedTokens > 0 ? { cachedTokens } : {}),\n }\n\n // Add completion token details\n // Flatten modality breakdown for candidates (output)\n const completionModalities = flattenModalityTokenCounts(\n usageMetadata.candidatesTokensDetails,\n )\n const thoughtsTokens = usageMetadata.thoughtsTokenCount\n\n const completionTokensDetails = {\n ...(hasModalityTokens(completionModalities) ? completionModalities : {}),\n // Map thoughtsTokenCount to reasoningTokens for consistency with OpenAI\n ...(thoughtsTokens !== undefined && thoughtsTokens > 0\n ? { reasoningTokens: thoughtsTokens }\n : {}),\n }\n\n // Add provider-specific details\n const providerDetails: GeminiProviderUsageDetails = {\n ...(usageMetadata.trafficType\n ? { trafficType: usageMetadata.trafficType }\n : {}),\n ...(usageMetadata.toolUsePromptTokenCount !== undefined &&\n usageMetadata.toolUsePromptTokenCount > 0\n ? { toolUsePromptTokenCount: usageMetadata.toolUsePromptTokenCount }\n : {}),\n ...(usageMetadata.toolUsePromptTokensDetails &&\n usageMetadata.toolUsePromptTokensDetails.length > 0\n ? {\n toolUsePromptTokensDetails:\n usageMetadata.toolUsePromptTokensDetails.map((item) => ({\n modality: item.modality || 'UNKNOWN',\n tokenCount: item.tokenCount ?? 0,\n })),\n }\n : {}),\n ...(usageMetadata.cacheTokensDetails &&\n usageMetadata.cacheTokensDetails.length > 0\n ? {\n cacheTokensDetails: usageMetadata.cacheTokensDetails.map((item) => ({\n modality: item.modality || 'UNKNOWN',\n tokenCount: item.tokenCount ?? 0,\n })),\n }\n : {}),\n }\n\n // Add prompt token details if available\n if (Object.keys(promptTokensDetails).length > 0) {\n result.promptTokensDetails = promptTokensDetails\n }\n // Add provider details if available\n if (Object.keys(providerDetails).length > 0) {\n result.providerUsageDetails = providerDetails\n }\n // Add completion token details if available\n if (Object.keys(completionTokensDetails).length > 0) {\n result.completionTokensDetails = completionTokensDetails\n }\n\n return result\n}\n"],"mappings":";;;;;;;AA6BA,SAAgB,2BACd,YACyB;CACzB,IAAI,CAAC,cAAc,WAAW,WAAW,GACvC,OAAO,CAAC;CAGV,MAAM,SAAkC,CAAC;CAEzC,KAAK,MAAM,QAAQ,YAAY;EAC7B,IAAI,CAAC,KAAK,YAAY,KAAK,eAAe,KAAA,GACxC;EAGF,MAAM,WAAW,KAAK,SAAS,YAAY;EAC3C,MAAM,QAAQ,KAAK;EAEnB,QAAQ,UAAR;GACE,KAAK;IACH,OAAO,cAAc,OAAO,cAAc,KAAK;IAC/C;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK
|
|
1
|
+
{"version":3,"file":"usage.js","names":[],"sources":["../../src/usage.ts"],"sourcesContent":["import { buildBaseUsage } from '@tanstack/ai'\nimport type { TokenUsage } from '@tanstack/ai'\nimport type {\n GenerateContentResponseUsageMetadata,\n ModalityTokenCount,\n} from '@google/genai'\n\n/**\n * Flattened modality token counts for normalized usage reporting.\n * Maps Gemini's ModalityTokenCount array to individual fields.\n */\nexport interface FlattenedModalityTokens {\n /** Text tokens */\n textTokens?: number\n /** Image tokens */\n imageTokens?: number\n /** Audio tokens */\n audioTokens?: number\n /** Video tokens */\n videoTokens?: number\n /** Document tokens (e.g. PDF inputs) */\n documentTokens?: number\n}\n\n/**\n * Flattens Gemini's ModalityTokenCount array into individual token fields.\n * Extracts TEXT, IMAGE, AUDIO, VIDEO, DOCUMENT modality counts into a\n * normalized structure.\n */\nexport function flattenModalityTokenCounts(\n modalities?: Array<ModalityTokenCount>,\n): FlattenedModalityTokens {\n if (!modalities || modalities.length === 0) {\n return {}\n }\n\n const result: FlattenedModalityTokens = {}\n\n for (const item of modalities) {\n if (!item.modality || item.tokenCount === undefined) {\n continue\n }\n\n const modality = item.modality.toUpperCase()\n const count = item.tokenCount\n\n switch (modality) {\n case 'TEXT':\n result.textTokens = (result.textTokens ?? 0) + count\n break\n case 'IMAGE':\n result.imageTokens = (result.imageTokens ?? 0) + count\n break\n case 'AUDIO':\n result.audioTokens = (result.audioTokens ?? 0) + count\n break\n case 'VIDEO':\n result.videoTokens = (result.videoTokens ?? 0) + count\n break\n case 'DOCUMENT':\n result.documentTokens = (result.documentTokens ?? 0) + count\n break\n }\n }\n\n return result\n}\n\n/**\n * Checks if a FlattenedModalityTokens object has any values set.\n */\nexport function hasModalityTokens(tokens: FlattenedModalityTokens): boolean {\n return (\n tokens.textTokens !== undefined ||\n tokens.imageTokens !== undefined ||\n tokens.audioTokens !== undefined ||\n tokens.videoTokens !== undefined ||\n tokens.documentTokens !== undefined\n )\n}\n\n/**\n * Gemini-specific provider usage details.\n * These fields are unique to Gemini and placed in providerUsageDetails.\n */\nexport type GeminiProviderUsageDetails = {\n /**\n * The traffic type for this request.\n * Can indicate whether request was handled by different service tiers.\n */\n trafficType?: string\n /**\n * Number of tokens in the results from tool executions,\n * which are provided back to the model as input.\n */\n toolUsePromptTokenCount?: number\n /**\n * Detailed breakdown by modality of the token counts from\n * the results of tool executions.\n */\n toolUsePromptTokensDetails?: Array<{\n modality: string\n tokenCount: number\n }>\n /**\n * Detailed breakdown of cache tokens by modality.\n * More granular than the normalized cachedTokens field.\n */\n cacheTokensDetails?: Array<{\n modality: string\n tokenCount: number\n }>\n}\n\n/**\n * Build normalized TokenUsage from Gemini's usageMetadata.\n * Handles modality breakdowns and thinking tokens. Returns `undefined` when the\n * provider reported no usage metadata, so callers omit the field rather than\n * fabricating zeroed totals.\n */\nexport function buildGeminiUsage(\n usageMetadata: GenerateContentResponseUsageMetadata | undefined | null,\n): TokenUsage<GeminiProviderUsageDetails> | undefined {\n if (!usageMetadata) return undefined\n\n const promptTokens = usageMetadata.promptTokenCount ?? 0\n const completionTokens = usageMetadata.candidatesTokenCount ?? 0\n\n const result = buildBaseUsage<GeminiProviderUsageDetails>({\n promptTokens: promptTokens,\n completionTokens: completionTokens,\n totalTokens:\n usageMetadata.totalTokenCount ?? promptTokens + completionTokens,\n })\n\n // Add prompt token details\n // Flatten modality breakdown for prompt\n const promptModalities = flattenModalityTokenCounts(\n usageMetadata.promptTokensDetails,\n )\n const cachedTokens = usageMetadata.cachedContentTokenCount\n\n const promptTokensDetails = {\n ...(hasModalityTokens(promptModalities) ? promptModalities : {}),\n ...(cachedTokens !== undefined && cachedTokens > 0 ? { cachedTokens } : {}),\n }\n\n // Add completion token details\n // Flatten modality breakdown for candidates (output)\n const completionModalities = flattenModalityTokenCounts(\n usageMetadata.candidatesTokensDetails,\n )\n const thoughtsTokens = usageMetadata.thoughtsTokenCount\n\n const completionTokensDetails = {\n ...(hasModalityTokens(completionModalities) ? completionModalities : {}),\n // Map thoughtsTokenCount to reasoningTokens for consistency with OpenAI\n ...(thoughtsTokens !== undefined && thoughtsTokens > 0\n ? { reasoningTokens: thoughtsTokens }\n : {}),\n }\n\n // Add provider-specific details\n const providerDetails: GeminiProviderUsageDetails = {\n ...(usageMetadata.trafficType\n ? { trafficType: usageMetadata.trafficType }\n : {}),\n ...(usageMetadata.toolUsePromptTokenCount !== undefined &&\n usageMetadata.toolUsePromptTokenCount > 0\n ? { toolUsePromptTokenCount: usageMetadata.toolUsePromptTokenCount }\n : {}),\n ...(usageMetadata.toolUsePromptTokensDetails &&\n usageMetadata.toolUsePromptTokensDetails.length > 0\n ? {\n toolUsePromptTokensDetails:\n usageMetadata.toolUsePromptTokensDetails.map((item) => ({\n modality: item.modality || 'UNKNOWN',\n tokenCount: item.tokenCount ?? 0,\n })),\n }\n : {}),\n ...(usageMetadata.cacheTokensDetails &&\n usageMetadata.cacheTokensDetails.length > 0\n ? {\n cacheTokensDetails: usageMetadata.cacheTokensDetails.map((item) => ({\n modality: item.modality || 'UNKNOWN',\n tokenCount: item.tokenCount ?? 0,\n })),\n }\n : {}),\n }\n\n // Add prompt token details if available\n if (Object.keys(promptTokensDetails).length > 0) {\n result.promptTokensDetails = promptTokensDetails\n }\n // Add provider details if available\n if (Object.keys(providerDetails).length > 0) {\n result.providerUsageDetails = providerDetails\n }\n // Add completion token details if available\n if (Object.keys(completionTokensDetails).length > 0) {\n result.completionTokensDetails = completionTokensDetails\n }\n\n return result\n}\n"],"mappings":";;;;;;;AA6BA,SAAgB,2BACd,YACyB;CACzB,IAAI,CAAC,cAAc,WAAW,WAAW,GACvC,OAAO,CAAC;CAGV,MAAM,SAAkC,CAAC;CAEzC,KAAK,MAAM,QAAQ,YAAY;EAC7B,IAAI,CAAC,KAAK,YAAY,KAAK,eAAe,KAAA,GACxC;EAGF,MAAM,WAAW,KAAK,SAAS,YAAY;EAC3C,MAAM,QAAQ,KAAK;EAEnB,QAAQ,UAAR;GACE,KAAK;IACH,OAAO,cAAc,OAAO,cAAc,KAAK;IAC/C;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK,YACH,OAAO,kBAAkB,OAAO,kBAAkB,KAAK;EAE3D;CACF;CAEA,OAAO;AACT;;;;AAKA,SAAgB,kBAAkB,QAA0C;CAC1E,OACE,OAAO,eAAe,KAAA,KACtB,OAAO,gBAAgB,KAAA,KACvB,OAAO,gBAAgB,KAAA,KACvB,OAAO,gBAAgB,KAAA,KACvB,OAAO,mBAAmB,KAAA;AAE9B;;;;;;;AAyCA,SAAgB,iBACd,eACoD;CACpD,IAAI,CAAC,eAAe,OAAO,KAAA;CAE3B,MAAM,eAAe,cAAc,oBAAoB;CACvD,MAAM,mBAAmB,cAAc,wBAAwB;CAE/D,MAAM,SAAS,eAA2C;EAC1C;EACI;EAClB,aACE,cAAc,mBAAmB,eAAe;CACpD,CAAC;CAID,MAAM,mBAAmB,2BACvB,cAAc,mBAChB;CACA,MAAM,eAAe,cAAc;CAEnC,MAAM,sBAAsB;EAC1B,GAAI,kBAAkB,gBAAgB,IAAI,mBAAmB,CAAC;EAC9D,GAAI,iBAAiB,KAAA,KAAa,eAAe,IAAI,EAAE,aAAa,IAAI,CAAC;CAC3E;CAIA,MAAM,uBAAuB,2BAC3B,cAAc,uBAChB;CACA,MAAM,iBAAiB,cAAc;CAErC,MAAM,0BAA0B;EAC9B,GAAI,kBAAkB,oBAAoB,IAAI,uBAAuB,CAAC;EAEtE,GAAI,mBAAmB,KAAA,KAAa,iBAAiB,IACjD,EAAE,iBAAiB,eAAe,IAClC,CAAC;CACP;CAGA,MAAM,kBAA8C;EAClD,GAAI,cAAc,cACd,EAAE,aAAa,cAAc,YAAY,IACzC,CAAC;EACL,GAAI,cAAc,4BAA4B,KAAA,KAC9C,cAAc,0BAA0B,IACpC,EAAE,yBAAyB,cAAc,wBAAwB,IACjE,CAAC;EACL,GAAI,cAAc,8BAClB,cAAc,2BAA2B,SAAS,IAC9C,EACE,4BACE,cAAc,2BAA2B,KAAK,UAAU;GACtD,UAAU,KAAK,YAAY;GAC3B,YAAY,KAAK,cAAc;EACjC,EAAE,EACN,IACA,CAAC;EACL,GAAI,cAAc,sBAClB,cAAc,mBAAmB,SAAS,IACtC,EACE,oBAAoB,cAAc,mBAAmB,KAAK,UAAU;GAClE,UAAU,KAAK,YAAY;GAC3B,YAAY,KAAK,cAAc;EACjC,EAAE,EACJ,IACA,CAAC;CACP;CAGA,IAAI,OAAO,KAAK,mBAAmB,CAAC,CAAC,SAAS,GAC5C,OAAO,sBAAsB;CAG/B,IAAI,OAAO,KAAK,eAAe,CAAC,CAAC,SAAS,GACxC,OAAO,uBAAuB;CAGhC,IAAI,OAAO,KAAK,uBAAuB,CAAC,CAAC,SAAS,GAChD,OAAO,0BAA0B;CAGnC,OAAO;AACT"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai-gemini",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.24.0",
|
|
4
4
|
"description": "Google Gemini adapter for TanStack AI chat, images, speech, audio generation, and structured outputs.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -59,13 +59,13 @@
|
|
|
59
59
|
"@tanstack/ai-utils": "^0.4.0"
|
|
60
60
|
},
|
|
61
61
|
"peerDependencies": {
|
|
62
|
-
"@tanstack/ai": "^0.
|
|
62
|
+
"@tanstack/ai": "^0.45.0"
|
|
63
63
|
},
|
|
64
64
|
"devDependencies": {
|
|
65
|
-
"@vitest/coverage-v8": "4.
|
|
66
|
-
"vite": "^8.1
|
|
65
|
+
"@vitest/coverage-v8": "4.1.10",
|
|
66
|
+
"vite": "^8.2.1",
|
|
67
67
|
"zod": "^4.2.0",
|
|
68
|
-
"@tanstack/ai": "0.
|
|
68
|
+
"@tanstack/ai": "0.45.0"
|
|
69
69
|
},
|
|
70
70
|
"scripts": {
|
|
71
71
|
"build": "vite build",
|
package/src/adapters/image.ts
CHANGED
|
@@ -7,18 +7,20 @@ import {
|
|
|
7
7
|
} from '../utils'
|
|
8
8
|
import { buildGeminiUsage } from '../usage'
|
|
9
9
|
import {
|
|
10
|
+
isGeminiNativeImageModel,
|
|
10
11
|
parseNativeImageSize,
|
|
11
12
|
sizeToAspectRatio,
|
|
12
13
|
validateImageSize,
|
|
13
14
|
validateNumberOfImages,
|
|
14
15
|
validatePrompt,
|
|
15
16
|
} from '../image/image-provider-options'
|
|
16
|
-
import type {
|
|
17
|
+
import type { GeminiImageModels } from '../model-meta'
|
|
17
18
|
import type {
|
|
19
|
+
GeminiAnyImageProviderOptions,
|
|
18
20
|
GeminiImageModelInputModalitiesByName,
|
|
19
21
|
GeminiImageModelProviderOptionsByName,
|
|
20
22
|
GeminiImageModelSizeByName,
|
|
21
|
-
|
|
23
|
+
GeminiNativeImageProviderOptions,
|
|
22
24
|
} from '../image/image-provider-options'
|
|
23
25
|
import type {
|
|
24
26
|
GeneratedImage,
|
|
@@ -35,6 +37,7 @@ import type {
|
|
|
35
37
|
GenerateImagesConfig,
|
|
36
38
|
GenerateImagesResponse,
|
|
37
39
|
GoogleGenAI,
|
|
40
|
+
ImageConfig,
|
|
38
41
|
Part,
|
|
39
42
|
} from '@google/genai'
|
|
40
43
|
import type { GeminiClientConfig } from '../utils/client'
|
|
@@ -45,7 +48,7 @@ import type { GeminiClientConfig } from '../utils/client'
|
|
|
45
48
|
export interface GeminiImageConfig extends GeminiClientConfig {}
|
|
46
49
|
|
|
47
50
|
/** Model type for Gemini Image */
|
|
48
|
-
export type GeminiImageModel =
|
|
51
|
+
export type GeminiImageModel = GeminiImageModels
|
|
49
52
|
|
|
50
53
|
/**
|
|
51
54
|
* Gemini Image Generation Adapter
|
|
@@ -65,7 +68,7 @@ export class GeminiImageAdapter<
|
|
|
65
68
|
TModel extends GeminiImageModel,
|
|
66
69
|
> extends BaseImageAdapter<
|
|
67
70
|
TModel,
|
|
68
|
-
|
|
71
|
+
GeminiAnyImageProviderOptions,
|
|
69
72
|
GeminiImageModelProviderOptionsByName,
|
|
70
73
|
GeminiImageModelSizeByName,
|
|
71
74
|
GeminiImageModelInputModalitiesByName
|
|
@@ -75,7 +78,7 @@ export class GeminiImageAdapter<
|
|
|
75
78
|
|
|
76
79
|
// Type-only property - never assigned at runtime
|
|
77
80
|
declare '~types': {
|
|
78
|
-
providerOptions:
|
|
81
|
+
providerOptions: GeminiAnyImageProviderOptions
|
|
79
82
|
modelProviderOptionsByName: GeminiImageModelProviderOptionsByName
|
|
80
83
|
modelSizeByName: GeminiImageModelSizeByName
|
|
81
84
|
modelInputModalitiesByName: GeminiImageModelInputModalitiesByName
|
|
@@ -89,7 +92,7 @@ export class GeminiImageAdapter<
|
|
|
89
92
|
}
|
|
90
93
|
|
|
91
94
|
async generateImages(
|
|
92
|
-
options: ImageGenerationOptions<
|
|
95
|
+
options: ImageGenerationOptions<GeminiAnyImageProviderOptions>,
|
|
93
96
|
): Promise<ImageGenerationResult> {
|
|
94
97
|
const { model, logger } = options
|
|
95
98
|
|
|
@@ -121,7 +124,7 @@ export class GeminiImageAdapter<
|
|
|
121
124
|
)
|
|
122
125
|
}
|
|
123
126
|
|
|
124
|
-
if (
|
|
127
|
+
if (isGeminiNativeImageModel(model)) {
|
|
125
128
|
return await this.generateWithGeminiApi(options, resolved)
|
|
126
129
|
}
|
|
127
130
|
|
|
@@ -155,29 +158,40 @@ export class GeminiImageAdapter<
|
|
|
155
158
|
}
|
|
156
159
|
}
|
|
157
160
|
|
|
158
|
-
private isGeminiImageModel(model: string): boolean {
|
|
159
|
-
return model.startsWith('gemini-')
|
|
160
|
-
}
|
|
161
|
-
|
|
162
161
|
private async generateWithGeminiApi(
|
|
163
|
-
options: ImageGenerationOptions<
|
|
162
|
+
options: ImageGenerationOptions<GeminiNativeImageProviderOptions>,
|
|
164
163
|
resolved: ResolvedMediaPrompt,
|
|
165
164
|
): Promise<ImageGenerationResult> {
|
|
166
165
|
const { model, size, numberOfImages, modelOptions } = options
|
|
167
166
|
|
|
168
167
|
const parsedSize = size ? parseNativeImageSize(size) : undefined
|
|
169
168
|
|
|
170
|
-
//
|
|
171
|
-
//
|
|
172
|
-
//
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
169
|
+
// The portable `size` option is the baseline; modelOptions.imageConfig is
|
|
170
|
+
// the provider escape hatch and wins per field, so a caller passing only
|
|
171
|
+
// `imageConfig.imageSize` keeps the aspectRatio derived from `size`.
|
|
172
|
+
const imageConfig: ImageConfig = {
|
|
173
|
+
...(parsedSize?.aspectRatio && { aspectRatio: parsedSize.aspectRatio }),
|
|
174
|
+
...(parsedSize?.resolution && { imageSize: parsedSize.resolution }),
|
|
175
|
+
...modelOptions?.imageConfig,
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
// Named picks, never a wholesale spread: the Imagen-shaped fields of
|
|
179
|
+
// GeminiImageProviderOptions (personGeneration, safetyFilterLevel,
|
|
180
|
+
// addWatermark, outputMimeType, …) are only valid on GenerateImagesConfig
|
|
181
|
+
// and would be rejected by generateContent. Picking by name means no
|
|
182
|
+
// Imagen field can reach this path even if one slips past the per-model
|
|
183
|
+
// provider-options map.
|
|
184
|
+
const nativeConfig: GenerateContentConfig = {
|
|
185
|
+
...(modelOptions?.seed !== undefined && { seed: modelOptions.seed }),
|
|
186
|
+
...(modelOptions?.safetySettings !== undefined && {
|
|
187
|
+
safetySettings: modelOptions.safetySettings,
|
|
188
|
+
}),
|
|
189
|
+
...(modelOptions?.thinkingConfig !== undefined && {
|
|
190
|
+
thinkingConfig: modelOptions.thinkingConfig,
|
|
191
|
+
}),
|
|
192
|
+
...(modelOptions?.systemInstruction !== undefined && {
|
|
193
|
+
systemInstruction: modelOptions.systemInstruction,
|
|
194
|
+
}),
|
|
181
195
|
}
|
|
182
196
|
|
|
183
197
|
const config: GenerateContentConfig = {
|
|
@@ -186,16 +200,7 @@ export class GeminiImageAdapter<
|
|
|
186
200
|
// IMPORTANT: responseModalities is a protected default — set it AFTER
|
|
187
201
|
// nativeConfig so nothing can silently disable image output.
|
|
188
202
|
responseModalities: ['TEXT', 'IMAGE'],
|
|
189
|
-
...(
|
|
190
|
-
imageConfig: {
|
|
191
|
-
...(parsedSize.aspectRatio && {
|
|
192
|
-
aspectRatio: parsedSize.aspectRatio,
|
|
193
|
-
}),
|
|
194
|
-
...(parsedSize.resolution && {
|
|
195
|
-
imageSize: parsedSize.resolution,
|
|
196
|
-
}),
|
|
197
|
-
},
|
|
198
|
-
}),
|
|
203
|
+
...(Object.keys(imageConfig).length > 0 && { imageConfig }),
|
|
199
204
|
}
|
|
200
205
|
|
|
201
206
|
const contents = this.buildContents(resolved, numberOfImages)
|
|
@@ -321,7 +326,7 @@ export class GeminiImageAdapter<
|
|
|
321
326
|
}
|
|
322
327
|
|
|
323
328
|
private buildImagenConfig(
|
|
324
|
-
options: ImageGenerationOptions<
|
|
329
|
+
options: ImageGenerationOptions<GeminiAnyImageProviderOptions>,
|
|
325
330
|
): GenerateImagesConfig {
|
|
326
331
|
const { size, numberOfImages, modelOptions } = options
|
|
327
332
|
|
|
@@ -329,11 +334,62 @@ export class GeminiImageAdapter<
|
|
|
329
334
|
// vendor `GenerateImagesConfig` fields are `field?: T` (no `| undefined`),
|
|
330
335
|
// so we can only assign the property when we actually have a value.
|
|
331
336
|
const sizeAspectRatio = size ? sizeToAspectRatio(size) : undefined
|
|
337
|
+
|
|
338
|
+
// Named picks, never a wholesale spread — the mirror image of the native
|
|
339
|
+
// path below. A native-only field (safetySettings, thinkingConfig,
|
|
340
|
+
// imageConfig, systemInstruction) belongs to GenerateContentConfig and is
|
|
341
|
+
// rejected by generateImages with 400 INVALID_ARGUMENT, so it must not be
|
|
342
|
+
// able to reach here even when the caller's `modelOptions` was typed
|
|
343
|
+
// against both shapes at once (e.g. an adapter inferred from a union of
|
|
344
|
+
// model names).
|
|
332
345
|
return {
|
|
333
346
|
numberOfImages: numberOfImages ?? 1,
|
|
334
|
-
// Map size to aspect ratio if provided
|
|
347
|
+
// Map size to aspect ratio if provided; modelOptions.aspectRatio,
|
|
348
|
+
// picked after it, overrides.
|
|
335
349
|
...(sizeAspectRatio !== undefined && { aspectRatio: sizeAspectRatio }),
|
|
336
|
-
...modelOptions
|
|
350
|
+
...(modelOptions?.aspectRatio !== undefined && {
|
|
351
|
+
aspectRatio: modelOptions.aspectRatio,
|
|
352
|
+
}),
|
|
353
|
+
...(modelOptions?.personGeneration !== undefined && {
|
|
354
|
+
personGeneration: modelOptions.personGeneration,
|
|
355
|
+
}),
|
|
356
|
+
...(modelOptions?.safetyFilterLevel !== undefined && {
|
|
357
|
+
safetyFilterLevel: modelOptions.safetyFilterLevel,
|
|
358
|
+
}),
|
|
359
|
+
...(modelOptions?.seed !== undefined && { seed: modelOptions.seed }),
|
|
360
|
+
...(modelOptions?.addWatermark !== undefined && {
|
|
361
|
+
addWatermark: modelOptions.addWatermark,
|
|
362
|
+
}),
|
|
363
|
+
...(modelOptions?.language !== undefined && {
|
|
364
|
+
language: modelOptions.language,
|
|
365
|
+
}),
|
|
366
|
+
...(modelOptions?.negativePrompt !== undefined && {
|
|
367
|
+
negativePrompt: modelOptions.negativePrompt,
|
|
368
|
+
}),
|
|
369
|
+
...(modelOptions?.outputMimeType !== undefined && {
|
|
370
|
+
outputMimeType: modelOptions.outputMimeType,
|
|
371
|
+
}),
|
|
372
|
+
...(modelOptions?.outputCompressionQuality !== undefined && {
|
|
373
|
+
outputCompressionQuality: modelOptions.outputCompressionQuality,
|
|
374
|
+
}),
|
|
375
|
+
...(modelOptions?.guidanceScale !== undefined && {
|
|
376
|
+
guidanceScale: modelOptions.guidanceScale,
|
|
377
|
+
}),
|
|
378
|
+
...(modelOptions?.enhancePrompt !== undefined && {
|
|
379
|
+
enhancePrompt: modelOptions.enhancePrompt,
|
|
380
|
+
}),
|
|
381
|
+
...(modelOptions?.includeSafetyAttributes !== undefined && {
|
|
382
|
+
includeSafetyAttributes: modelOptions.includeSafetyAttributes,
|
|
383
|
+
}),
|
|
384
|
+
...(modelOptions?.includeRaiReason !== undefined && {
|
|
385
|
+
includeRaiReason: modelOptions.includeRaiReason,
|
|
386
|
+
}),
|
|
387
|
+
...(modelOptions?.outputGcsUri !== undefined && {
|
|
388
|
+
outputGcsUri: modelOptions.outputGcsUri,
|
|
389
|
+
}),
|
|
390
|
+
...(modelOptions?.labels !== undefined && {
|
|
391
|
+
labels: modelOptions.labels,
|
|
392
|
+
}),
|
|
337
393
|
}
|
|
338
394
|
}
|
|
339
395
|
|
|
@@ -392,6 +448,18 @@ export class GeminiImageAdapter<
|
|
|
392
448
|
}
|
|
393
449
|
}
|
|
394
450
|
|
|
451
|
+
/** @deprecated Shut down 2026-06-25. Use `gemini-3.1-flash-image`. */
|
|
452
|
+
export function createGeminiImage(
|
|
453
|
+
model: 'gemini-3.1-flash-image-preview',
|
|
454
|
+
apiKey: string,
|
|
455
|
+
config?: Omit<GeminiImageConfig, 'apiKey'>,
|
|
456
|
+
): GeminiImageAdapter<'gemini-3.1-flash-image-preview'>
|
|
457
|
+
/** @deprecated Shut down 2026-06-25. Use `gemini-3-pro-image`. */
|
|
458
|
+
export function createGeminiImage(
|
|
459
|
+
model: 'gemini-3-pro-image-preview',
|
|
460
|
+
apiKey: string,
|
|
461
|
+
config?: Omit<GeminiImageConfig, 'apiKey'>,
|
|
462
|
+
): GeminiImageAdapter<'gemini-3-pro-image-preview'>
|
|
395
463
|
/**
|
|
396
464
|
* Creates a Gemini image adapter with explicit API key.
|
|
397
465
|
* Type resolution happens here at the call site.
|
|
@@ -411,6 +479,11 @@ export class GeminiImageAdapter<
|
|
|
411
479
|
* });
|
|
412
480
|
* ```
|
|
413
481
|
*/
|
|
482
|
+
export function createGeminiImage<TModel extends GeminiImageModel>(
|
|
483
|
+
model: TModel,
|
|
484
|
+
apiKey: string,
|
|
485
|
+
config?: Omit<GeminiImageConfig, 'apiKey'>,
|
|
486
|
+
): GeminiImageAdapter<TModel>
|
|
414
487
|
export function createGeminiImage<TModel extends GeminiImageModel>(
|
|
415
488
|
model: TModel,
|
|
416
489
|
apiKey: string,
|
|
@@ -419,6 +492,16 @@ export function createGeminiImage<TModel extends GeminiImageModel>(
|
|
|
419
492
|
return new GeminiImageAdapter({ apiKey, ...config }, model)
|
|
420
493
|
}
|
|
421
494
|
|
|
495
|
+
/** @deprecated Shut down 2026-06-25. Use `gemini-3.1-flash-image`. */
|
|
496
|
+
export function geminiImage(
|
|
497
|
+
model: 'gemini-3.1-flash-image-preview',
|
|
498
|
+
config?: Omit<GeminiImageConfig, 'apiKey'>,
|
|
499
|
+
): GeminiImageAdapter<'gemini-3.1-flash-image-preview'>
|
|
500
|
+
/** @deprecated Shut down 2026-06-25. Use `gemini-3-pro-image`. */
|
|
501
|
+
export function geminiImage(
|
|
502
|
+
model: 'gemini-3-pro-image-preview',
|
|
503
|
+
config?: Omit<GeminiImageConfig, 'apiKey'>,
|
|
504
|
+
): GeminiImageAdapter<'gemini-3-pro-image-preview'>
|
|
422
505
|
/**
|
|
423
506
|
* Creates a Gemini image adapter with automatic API key detection from environment variables.
|
|
424
507
|
* Type resolution happens here at the call site.
|
|
@@ -443,6 +526,10 @@ export function createGeminiImage<TModel extends GeminiImageModel>(
|
|
|
443
526
|
* });
|
|
444
527
|
* ```
|
|
445
528
|
*/
|
|
529
|
+
export function geminiImage<TModel extends GeminiImageModel>(
|
|
530
|
+
model: TModel,
|
|
531
|
+
config?: Omit<GeminiImageConfig, 'apiKey'>,
|
|
532
|
+
): GeminiImageAdapter<TModel>
|
|
446
533
|
export function geminiImage<TModel extends GeminiImageModel>(
|
|
447
534
|
model: TModel,
|
|
448
535
|
config?: Omit<GeminiImageConfig, 'apiKey'>,
|
|
@@ -1,12 +1,24 @@
|
|
|
1
1
|
import type { GeminiImageModels } from '../model-meta'
|
|
2
2
|
import type {
|
|
3
|
+
ContentUnion,
|
|
4
|
+
ImageConfig,
|
|
3
5
|
ImagePromptLanguage,
|
|
4
6
|
PersonGeneration,
|
|
5
7
|
SafetyFilterLevel,
|
|
8
|
+
SafetySetting,
|
|
9
|
+
ThinkingConfig,
|
|
6
10
|
} from '@google/genai'
|
|
7
11
|
|
|
8
12
|
// Re-export SDK types so users can use them directly
|
|
9
|
-
export type {
|
|
13
|
+
export type {
|
|
14
|
+
ContentUnion,
|
|
15
|
+
ImageConfig,
|
|
16
|
+
ImagePromptLanguage,
|
|
17
|
+
PersonGeneration,
|
|
18
|
+
SafetyFilterLevel,
|
|
19
|
+
SafetySetting,
|
|
20
|
+
ThinkingConfig,
|
|
21
|
+
}
|
|
10
22
|
|
|
11
23
|
/**
|
|
12
24
|
* Gemini Imagen aspect ratio options
|
|
@@ -121,11 +133,80 @@ export interface GeminiImageProviderOptions {
|
|
|
121
133
|
}
|
|
122
134
|
|
|
123
135
|
/**
|
|
124
|
-
*
|
|
125
|
-
*
|
|
136
|
+
* Provider options for Gemini native image models (Nano Banana and friends).
|
|
137
|
+
*
|
|
138
|
+
* These models are served by `generateContent`, not `generateImages`, so they
|
|
139
|
+
* are configured by @google/genai's `GenerateContentConfig` — a different
|
|
140
|
+
* shape from the Imagen-only {@link GeminiImageProviderOptions} above. Only
|
|
141
|
+
* the `GenerateContentConfig` fields with clear image-generation semantics are
|
|
142
|
+
* surfaced; sampling knobs (`temperature`, `topK`, …) and chat-only plumbing
|
|
143
|
+
* (`tools`, `responseSchema`, …) are deliberately left out.
|
|
144
|
+
*
|
|
145
|
+
* `responseModalities` is intentionally absent: the adapter always requests
|
|
146
|
+
* `['TEXT', 'IMAGE']`, and letting a caller override it would silently disable
|
|
147
|
+
* image output on an image-generation call.
|
|
148
|
+
*/
|
|
149
|
+
export interface GeminiNativeImageProviderOptions {
|
|
150
|
+
/**
|
|
151
|
+
* Optional seed for reproducible image generation
|
|
152
|
+
* When the same seed is used with the same prompt and settings,
|
|
153
|
+
* you should get similar (though not identical) results
|
|
154
|
+
*/
|
|
155
|
+
seed?: number
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Per-category safety thresholds applied to the request
|
|
159
|
+
* Each entry pairs a HarmCategory with a HarmBlockThreshold
|
|
160
|
+
*/
|
|
161
|
+
safetySettings?: Array<SafetySetting>
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Controls the model's internal reasoning before it emits an image
|
|
165
|
+
* Use to raise or disable the thinking budget on models that support it
|
|
166
|
+
*/
|
|
167
|
+
thinkingConfig?: ThinkingConfig
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Native image output controls. Merged over the values derived from the
|
|
171
|
+
* portable `size` option, so fields set here win per field while the rest
|
|
172
|
+
* of `size` is preserved.
|
|
173
|
+
*
|
|
174
|
+
* Only `aspectRatio` and `imageSize` are accepted on the Gemini Developer
|
|
175
|
+
* API. Other SDK `ImageConfig` keys throw on this surface.
|
|
176
|
+
*/
|
|
177
|
+
imageConfig?: GeminiNativeImageConfig
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* System-level instructions that steer the model for the whole request,
|
|
181
|
+
* e.g. a house art direction applied on top of the per-call prompt
|
|
182
|
+
*/
|
|
183
|
+
systemInstruction?: ContentUnion
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* Every provider-option field this adapter understands, across both API
|
|
188
|
+
* paths. Used as the adapter's base (model-agnostic) option type; the
|
|
189
|
+
* per-model map below is what narrows a given model to the half that
|
|
190
|
+
* actually applies to it.
|
|
191
|
+
*/
|
|
192
|
+
export type GeminiAnyImageProviderOptions = GeminiImageProviderOptions &
|
|
193
|
+
GeminiNativeImageProviderOptions
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* Model-specific provider options mapping.
|
|
197
|
+
* Gemini native image models go through `generateContent` and take
|
|
198
|
+
* `GenerateContentConfig` fields; Imagen models go through `generateImages`
|
|
199
|
+
* and take `GenerateImagesConfig` fields. Mirrors the native/Imagen split in
|
|
200
|
+
* {@link GeminiImageModelSizeByName} and
|
|
201
|
+
* {@link GeminiImageModelInputModalitiesByName}.
|
|
126
202
|
*/
|
|
127
203
|
export type GeminiImageModelProviderOptionsByName = {
|
|
128
|
-
[K in
|
|
204
|
+
[K in GeminiNativeImageModels]: GeminiNativeImageProviderOptions
|
|
205
|
+
} & {
|
|
206
|
+
[K in Exclude<
|
|
207
|
+
GeminiImageModels,
|
|
208
|
+
GeminiNativeImageModels
|
|
209
|
+
>]: GeminiImageProviderOptions
|
|
129
210
|
}
|
|
130
211
|
|
|
131
212
|
/**
|
|
@@ -145,47 +226,157 @@ export type GeminiImageSize =
|
|
|
145
226
|
| '1080x1920'
|
|
146
227
|
|
|
147
228
|
/**
|
|
148
|
-
*
|
|
149
|
-
*
|
|
229
|
+
* The ten aspect ratios every Gemini native image model accepts.
|
|
230
|
+
*
|
|
231
|
+
* Note `9:21` is deliberately absent: it exists only on Vertex / Cloud and is
|
|
232
|
+
* rejected by the Gemini API (`generateContent`), which is the surface this
|
|
233
|
+
* adapter targets.
|
|
234
|
+
*
|
|
235
|
+
* @see https://ai.google.dev/gemini-api/docs/image-generation
|
|
150
236
|
*/
|
|
151
|
-
export type
|
|
237
|
+
export type GeminiStandardImageAspectRatio =
|
|
152
238
|
| '1:1'
|
|
153
239
|
| '2:3'
|
|
154
240
|
| '3:2'
|
|
155
241
|
| '3:4'
|
|
156
242
|
| '4:3'
|
|
243
|
+
| '4:5'
|
|
244
|
+
| '5:4'
|
|
157
245
|
| '9:16'
|
|
158
246
|
| '16:9'
|
|
159
247
|
| '21:9'
|
|
160
248
|
|
|
161
249
|
/**
|
|
162
|
-
*
|
|
163
|
-
*
|
|
250
|
+
* The ten standard ratios plus the four extreme banner/strip ratios that only
|
|
251
|
+
* the Gemini 3.1 Flash Image models accept — 14 values, matching the
|
|
252
|
+
* `generateContent` `ImageConfig.aspectRatio` field union.
|
|
253
|
+
*
|
|
254
|
+
* @see https://ai.google.dev/api/generate-content
|
|
255
|
+
*/
|
|
256
|
+
export type GeminiExtendedImageAspectRatio =
|
|
257
|
+
| GeminiStandardImageAspectRatio
|
|
258
|
+
| '1:4'
|
|
259
|
+
| '4:1'
|
|
260
|
+
| '1:8'
|
|
261
|
+
| '8:1'
|
|
262
|
+
|
|
263
|
+
/**
|
|
264
|
+
* Sizes for `gemini-3.1-flash-image` (and its shut-down `-preview` alias):
|
|
265
|
+
* all 14 aspect ratios at 512 / 1K / 2K / 4K. `512` is the wire token for the
|
|
266
|
+
* 0.5K tier — not `512px`, and the `K` is case-sensitive (`1k` is rejected).
|
|
267
|
+
*/
|
|
268
|
+
export type Gemini31FlashImageSize =
|
|
269
|
+
`${GeminiExtendedImageAspectRatio}_${'512' | '1K' | '2K' | '4K'}`
|
|
270
|
+
|
|
271
|
+
/**
|
|
272
|
+
* Sizes for `gemini-3.1-flash-lite-image`: all 14 aspect ratios, 1K only.
|
|
273
|
+
* 2K and 4K are unsupported on this model.
|
|
274
|
+
*
|
|
275
|
+
* The four banner ratios (`1:4`, `4:1`, `1:8`, `8:1`) come from the Cloud
|
|
276
|
+
* model page. The Gemini API page states a count of 14 but does not list them.
|
|
277
|
+
*
|
|
278
|
+
* @see https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/gemini/3-1-flash-lite-image
|
|
279
|
+
* @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image
|
|
280
|
+
*/
|
|
281
|
+
export type Gemini31FlashLiteImageSize = `${GeminiExtendedImageAspectRatio}_1K`
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Sizes for `gemini-3-pro-image` (and its shut-down `-preview` alias): the ten
|
|
285
|
+
* standard aspect ratios at 1K / 2K / 4K. Pro has no 512 tier and none of the
|
|
286
|
+
* extreme banner ratios on the Gemini API.
|
|
287
|
+
*/
|
|
288
|
+
export type Gemini3ProImageSize =
|
|
289
|
+
`${GeminiStandardImageAspectRatio}_${'1K' | '2K' | '4K'}`
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* Sizes for `gemini-2.5-flash-image`: a bare aspect ratio with no resolution
|
|
293
|
+
* suffix, e.g. `'16:9'`. Google documents no `image_size` value or default for
|
|
294
|
+
* this model — it emits a single fixed 1024px-class output — so the adapter
|
|
295
|
+
* sends `imageConfig.aspectRatio` and omits `imageSize` entirely rather than
|
|
296
|
+
* guessing a tier the API never documented.
|
|
164
297
|
*/
|
|
165
|
-
export type
|
|
298
|
+
export type Gemini25FlashImageSize = GeminiStandardImageAspectRatio
|
|
166
299
|
|
|
167
300
|
/**
|
|
168
|
-
*
|
|
301
|
+
* `imageConfig` fields the Gemini Developer API accepts on `generateContent`.
|
|
302
|
+
* Other `@google/genai` `ImageConfig` keys (`personGeneration`,
|
|
303
|
+
* `outputMimeType`, and more) throw on this surface.
|
|
304
|
+
*/
|
|
305
|
+
export type GeminiNativeImageConfig = {
|
|
306
|
+
aspectRatio?: GeminiExtendedImageAspectRatio
|
|
307
|
+
imageSize?: '512' | '1K' | '2K' | '4K'
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/**
|
|
311
|
+
* Any size accepted by any Gemini native image model. Prefer the per-model
|
|
312
|
+
* narrowing in {@link GeminiImageModelSizeByName} — this union is the widest
|
|
313
|
+
* possible set and accepts combinations no single model supports.
|
|
169
314
|
*/
|
|
170
315
|
export type GeminiNativeImageSize =
|
|
171
|
-
|
|
316
|
+
| Gemini31FlashImageSize
|
|
317
|
+
| Gemini31FlashLiteImageSize
|
|
318
|
+
| Gemini3ProImageSize
|
|
319
|
+
| Gemini25FlashImageSize
|
|
172
320
|
|
|
173
321
|
/**
|
|
174
322
|
* Gemini native image models that use the generateContent API path.
|
|
175
|
-
* These models
|
|
323
|
+
* These models take an aspect-ratio-based size rather than Imagen's
|
|
324
|
+
* WIDTHxHEIGHT pixel strings.
|
|
325
|
+
*
|
|
326
|
+
* This array is the single source of truth for the native/Imagen split: the
|
|
327
|
+
* `GeminiNativeImageModels` union and the per-model option/size/modality maps
|
|
328
|
+
* all derive from it. The `satisfies` clause makes a typo (or a name that
|
|
329
|
+
* is not a known image model) a build error rather than a phantom key on every
|
|
330
|
+
* per-model map.
|
|
331
|
+
*
|
|
332
|
+
* It is also the single source of truth for the adapter's runtime routing
|
|
333
|
+
* — see {@link isGeminiNativeImageModel}. Adding a new `gemini-*` image model
|
|
334
|
+
* means adding it here as well as to `GEMINI_IMAGE_MODELS` in model-meta.
|
|
335
|
+
* Until it is listed here it routes to the Imagen API instead and fails
|
|
336
|
+
* loudly on the first call, rather than silently taking the wrong option
|
|
337
|
+
* shape.
|
|
176
338
|
*/
|
|
339
|
+
export const GEMINI_NATIVE_IMAGE_MODELS = [
|
|
340
|
+
'gemini-3.1-flash-image',
|
|
341
|
+
'gemini-3.1-flash-image-preview',
|
|
342
|
+
'gemini-3.1-flash-lite-image',
|
|
343
|
+
'gemini-3-pro-image',
|
|
344
|
+
'gemini-3-pro-image-preview',
|
|
345
|
+
'gemini-2.5-flash-image',
|
|
346
|
+
] as const satisfies ReadonlyArray<GeminiImageModels>
|
|
347
|
+
|
|
177
348
|
export type GeminiNativeImageModels =
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
349
|
+
(typeof GEMINI_NATIVE_IMAGE_MODELS)[number]
|
|
350
|
+
|
|
351
|
+
const NATIVE_IMAGE_MODEL_NAMES: ReadonlySet<string> = new Set(
|
|
352
|
+
GEMINI_NATIVE_IMAGE_MODELS,
|
|
353
|
+
)
|
|
182
354
|
|
|
183
355
|
/**
|
|
184
|
-
*
|
|
185
|
-
* Gemini
|
|
356
|
+
* Runtime counterpart to {@link GeminiNativeImageModels} — decides which of
|
|
357
|
+
* the two Gemini image APIs a model goes to.
|
|
358
|
+
*
|
|
359
|
+
* Membership in {@link GEMINI_NATIVE_IMAGE_MODELS}, not a `gemini-` prefix
|
|
360
|
+
* test, so the runtime route and the type-level split cannot drift apart. An
|
|
361
|
+
* id this package does not know about reaches the Imagen endpoint and fails
|
|
362
|
+
* there, which is the intended signal to add the model here rather than to
|
|
363
|
+
* have it silently take the native path with Imagen-shaped option types.
|
|
364
|
+
*/
|
|
365
|
+
export function isGeminiNativeImageModel(model: string): boolean {
|
|
366
|
+
return NATIVE_IMAGE_MODEL_NAMES.has(model)
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
/**
|
|
370
|
+
* Model-specific size options mapping. Each native model gets its own ratio ×
|
|
371
|
+
* resolution set (they genuinely differ); Imagen models use pixel sizes.
|
|
186
372
|
*/
|
|
187
373
|
export type GeminiImageModelSizeByName = {
|
|
188
|
-
|
|
374
|
+
'gemini-3.1-flash-image': Gemini31FlashImageSize
|
|
375
|
+
'gemini-3.1-flash-image-preview': Gemini31FlashImageSize
|
|
376
|
+
'gemini-3.1-flash-lite-image': Gemini31FlashLiteImageSize
|
|
377
|
+
'gemini-3-pro-image': Gemini3ProImageSize
|
|
378
|
+
'gemini-3-pro-image-preview': Gemini3ProImageSize
|
|
379
|
+
'gemini-2.5-flash-image': Gemini25FlashImageSize
|
|
189
380
|
} & {
|
|
190
381
|
[K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: GeminiImageSize
|
|
191
382
|
}
|
|
@@ -309,13 +500,23 @@ export function validatePrompt(options: {
|
|
|
309
500
|
|
|
310
501
|
/**
|
|
311
502
|
* Parses a Gemini native image size string into its components.
|
|
312
|
-
*
|
|
503
|
+
*
|
|
504
|
+
* Format: `"aspectRatio_resolution"`, e.g. `"16:9_4K"` →
|
|
505
|
+
* `{ aspectRatio: "16:9", resolution: "4K" }`.
|
|
506
|
+
*
|
|
507
|
+
* The resolution suffix is optional: `gemini-2.5-flash-image` takes a bare
|
|
508
|
+
* aspect ratio (`"16:9"` → `{ aspectRatio: "16:9" }`) because Google documents
|
|
509
|
+
* no `image_size` for it, and the caller must then omit `imageSize` from the
|
|
510
|
+
* request rather than substituting a default.
|
|
313
511
|
*/
|
|
314
512
|
export function parseNativeImageSize(
|
|
315
513
|
size: string,
|
|
316
|
-
): { aspectRatio: string; resolution
|
|
317
|
-
const match = size.match(/^(\d+:\d+)_(.+)
|
|
514
|
+
): { aspectRatio: string; resolution?: string } | undefined {
|
|
515
|
+
const match = size.match(/^(\d+:\d+)(?:_(.+))?$/)
|
|
318
516
|
const [, aspectRatio, resolution] = match ?? []
|
|
319
|
-
if (aspectRatio === undefined
|
|
320
|
-
return {
|
|
517
|
+
if (aspectRatio === undefined) return undefined
|
|
518
|
+
return {
|
|
519
|
+
aspectRatio,
|
|
520
|
+
...(resolution !== undefined && { resolution }),
|
|
521
|
+
}
|
|
321
522
|
}
|