@tanstack/ai-gemini 0.23.0 → 0.24.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1 +1 @@
1
- {"version":3,"file":"usage.js","names":[],"sources":["../../src/usage.ts"],"sourcesContent":["import { buildBaseUsage } from '@tanstack/ai'\nimport type { TokenUsage } from '@tanstack/ai'\nimport type {\n GenerateContentResponseUsageMetadata,\n ModalityTokenCount,\n} from '@google/genai'\n\n/**\n * Flattened modality token counts for normalized usage reporting.\n * Maps Gemini's ModalityTokenCount array to individual fields.\n */\nexport interface FlattenedModalityTokens {\n /** Text tokens */\n textTokens?: number\n /** Image tokens */\n imageTokens?: number\n /** Audio tokens */\n audioTokens?: number\n /** Video tokens */\n videoTokens?: number\n /** Document tokens (e.g. PDF inputs) */\n documentTokens?: number\n}\n\n/**\n * Flattens Gemini's ModalityTokenCount array into individual token fields.\n * Extracts TEXT, IMAGE, AUDIO, VIDEO, DOCUMENT modality counts into a\n * normalized structure.\n */\nexport function flattenModalityTokenCounts(\n modalities?: Array<ModalityTokenCount>,\n): FlattenedModalityTokens {\n if (!modalities || modalities.length === 0) {\n return {}\n }\n\n const result: FlattenedModalityTokens = {}\n\n for (const item of modalities) {\n if (!item.modality || item.tokenCount === undefined) {\n continue\n }\n\n const modality = item.modality.toUpperCase()\n const count = item.tokenCount\n\n switch (modality) {\n case 'TEXT':\n result.textTokens = (result.textTokens ?? 0) + count\n break\n case 'IMAGE':\n result.imageTokens = (result.imageTokens ?? 0) + count\n break\n case 'AUDIO':\n result.audioTokens = (result.audioTokens ?? 0) + count\n break\n case 'VIDEO':\n result.videoTokens = (result.videoTokens ?? 0) + count\n break\n case 'DOCUMENT':\n result.documentTokens = (result.documentTokens ?? 0) + count\n break\n }\n }\n\n return result\n}\n\n/**\n * Checks if a FlattenedModalityTokens object has any values set.\n */\nexport function hasModalityTokens(tokens: FlattenedModalityTokens): boolean {\n return (\n tokens.textTokens !== undefined ||\n tokens.imageTokens !== undefined ||\n tokens.audioTokens !== undefined ||\n tokens.videoTokens !== undefined ||\n tokens.documentTokens !== undefined\n )\n}\n\n/**\n * Gemini-specific provider usage details.\n * These fields are unique to Gemini and placed in providerUsageDetails.\n */\nexport type GeminiProviderUsageDetails = {\n /**\n * The traffic type for this request.\n * Can indicate whether request was handled by different service tiers.\n */\n trafficType?: string\n /**\n * Number of tokens in the results from tool executions,\n * which are provided back to the model as input.\n */\n toolUsePromptTokenCount?: number\n /**\n * Detailed breakdown by modality of the token counts from\n * the results of tool executions.\n */\n toolUsePromptTokensDetails?: Array<{\n modality: string\n tokenCount: number\n }>\n /**\n * Detailed breakdown of cache tokens by modality.\n * More granular than the normalized cachedTokens field.\n */\n cacheTokensDetails?: Array<{\n modality: string\n tokenCount: number\n }>\n}\n\n/**\n * Build normalized TokenUsage from Gemini's usageMetadata.\n * Handles modality breakdowns and thinking tokens. Returns `undefined` when the\n * provider reported no usage metadata, so callers omit the field rather than\n * fabricating zeroed totals.\n */\nexport function buildGeminiUsage(\n usageMetadata: GenerateContentResponseUsageMetadata | undefined | null,\n): TokenUsage<GeminiProviderUsageDetails> | undefined {\n if (!usageMetadata) return undefined\n\n const promptTokens = usageMetadata.promptTokenCount ?? 0\n const completionTokens = usageMetadata.candidatesTokenCount ?? 0\n\n const result = buildBaseUsage<GeminiProviderUsageDetails>({\n promptTokens: promptTokens,\n completionTokens: completionTokens,\n totalTokens:\n usageMetadata.totalTokenCount ?? promptTokens + completionTokens,\n })\n\n // Add prompt token details\n // Flatten modality breakdown for prompt\n const promptModalities = flattenModalityTokenCounts(\n usageMetadata.promptTokensDetails,\n )\n const cachedTokens = usageMetadata.cachedContentTokenCount\n\n const promptTokensDetails = {\n ...(hasModalityTokens(promptModalities) ? promptModalities : {}),\n ...(cachedTokens !== undefined && cachedTokens > 0 ? { cachedTokens } : {}),\n }\n\n // Add completion token details\n // Flatten modality breakdown for candidates (output)\n const completionModalities = flattenModalityTokenCounts(\n usageMetadata.candidatesTokensDetails,\n )\n const thoughtsTokens = usageMetadata.thoughtsTokenCount\n\n const completionTokensDetails = {\n ...(hasModalityTokens(completionModalities) ? completionModalities : {}),\n // Map thoughtsTokenCount to reasoningTokens for consistency with OpenAI\n ...(thoughtsTokens !== undefined && thoughtsTokens > 0\n ? { reasoningTokens: thoughtsTokens }\n : {}),\n }\n\n // Add provider-specific details\n const providerDetails: GeminiProviderUsageDetails = {\n ...(usageMetadata.trafficType\n ? { trafficType: usageMetadata.trafficType }\n : {}),\n ...(usageMetadata.toolUsePromptTokenCount !== undefined &&\n usageMetadata.toolUsePromptTokenCount > 0\n ? { toolUsePromptTokenCount: usageMetadata.toolUsePromptTokenCount }\n : {}),\n ...(usageMetadata.toolUsePromptTokensDetails &&\n usageMetadata.toolUsePromptTokensDetails.length > 0\n ? {\n toolUsePromptTokensDetails:\n usageMetadata.toolUsePromptTokensDetails.map((item) => ({\n modality: item.modality || 'UNKNOWN',\n tokenCount: item.tokenCount ?? 0,\n })),\n }\n : {}),\n ...(usageMetadata.cacheTokensDetails &&\n usageMetadata.cacheTokensDetails.length > 0\n ? {\n cacheTokensDetails: usageMetadata.cacheTokensDetails.map((item) => ({\n modality: item.modality || 'UNKNOWN',\n tokenCount: item.tokenCount ?? 0,\n })),\n }\n : {}),\n }\n\n // Add prompt token details if available\n if (Object.keys(promptTokensDetails).length > 0) {\n result.promptTokensDetails = promptTokensDetails\n }\n // Add provider details if available\n if (Object.keys(providerDetails).length > 0) {\n result.providerUsageDetails = providerDetails\n }\n // Add completion token details if available\n if (Object.keys(completionTokensDetails).length > 0) {\n result.completionTokensDetails = completionTokensDetails\n }\n\n return result\n}\n"],"mappings":";;;;;;;AA6BA,SAAgB,2BACd,YACyB;CACzB,IAAI,CAAC,cAAc,WAAW,WAAW,GACvC,OAAO,CAAC;CAGV,MAAM,SAAkC,CAAC;CAEzC,KAAK,MAAM,QAAQ,YAAY;EAC7B,IAAI,CAAC,KAAK,YAAY,KAAK,eAAe,KAAA,GACxC;EAGF,MAAM,WAAW,KAAK,SAAS,YAAY;EAC3C,MAAM,QAAQ,KAAK;EAEnB,QAAQ,UAAR;GACE,KAAK;IACH,OAAO,cAAc,OAAO,cAAc,KAAK;IAC/C;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,kBAAkB,OAAO,kBAAkB,KAAK;IACvD;EACJ;CACF;CAEA,OAAO;AACT;;;;AAKA,SAAgB,kBAAkB,QAA0C;CAC1E,OACE,OAAO,eAAe,KAAA,KACtB,OAAO,gBAAgB,KAAA,KACvB,OAAO,gBAAgB,KAAA,KACvB,OAAO,gBAAgB,KAAA,KACvB,OAAO,mBAAmB,KAAA;AAE9B;;;;;;;AAyCA,SAAgB,iBACd,eACoD;CACpD,IAAI,CAAC,eAAe,OAAO,KAAA;CAE3B,MAAM,eAAe,cAAc,oBAAoB;CACvD,MAAM,mBAAmB,cAAc,wBAAwB;CAE/D,MAAM,SAAS,eAA2C;EAC1C;EACI;EAClB,aACE,cAAc,mBAAmB,eAAe;CACpD,CAAC;CAID,MAAM,mBAAmB,2BACvB,cAAc,mBAChB;CACA,MAAM,eAAe,cAAc;CAEnC,MAAM,sBAAsB;EAC1B,GAAI,kBAAkB,gBAAgB,IAAI,mBAAmB,CAAC;EAC9D,GAAI,iBAAiB,KAAA,KAAa,eAAe,IAAI,EAAE,aAAa,IAAI,CAAC;CAC3E;CAIA,MAAM,uBAAuB,2BAC3B,cAAc,uBAChB;CACA,MAAM,iBAAiB,cAAc;CAErC,MAAM,0BAA0B;EAC9B,GAAI,kBAAkB,oBAAoB,IAAI,uBAAuB,CAAC;EAEtE,GAAI,mBAAmB,KAAA,KAAa,iBAAiB,IACjD,EAAE,iBAAiB,eAAe,IAClC,CAAC;CACP;CAGA,MAAM,kBAA8C;EAClD,GAAI,cAAc,cACd,EAAE,aAAa,cAAc,YAAY,IACzC,CAAC;EACL,GAAI,cAAc,4BAA4B,KAAA,KAC9C,cAAc,0BAA0B,IACpC,EAAE,yBAAyB,cAAc,wBAAwB,IACjE,CAAC;EACL,GAAI,cAAc,8BAClB,cAAc,2BAA2B,SAAS,IAC9C,EACE,4BACE,cAAc,2BAA2B,KAAK,UAAU;GACtD,UAAU,KAAK,YAAY;GAC3B,YAAY,KAAK,cAAc;EACjC,EAAE,EACN,IACA,CAAC;EACL,GAAI,cAAc,sBAClB,cAAc,mBAAmB,SAAS,IACtC,EACE,oBAAoB,cAAc,mBAAmB,KAAK,UAAU;GAClE,UAAU,KAAK,YAAY;GAC3B,YAAY,KAAK,cAAc;EACjC,EAAE,EACJ,IACA,CAAC;CACP;CAGA,IAAI,OAAO,KAAK,mBAAmB,CAAC,CAAC,SAAS,GAC5C,OAAO,sBAAsB;CAG/B,IAAI,OAAO,KAAK,eAAe,CAAC,CAAC,SAAS,GACxC,OAAO,uBAAuB;CAGhC,IAAI,OAAO,KAAK,uBAAuB,CAAC,CAAC,SAAS,GAChD,OAAO,0BAA0B;CAGnC,OAAO;AACT"}
1
+ {"version":3,"file":"usage.js","names":[],"sources":["../../src/usage.ts"],"sourcesContent":["import { buildBaseUsage } from '@tanstack/ai'\nimport type { TokenUsage } from '@tanstack/ai'\nimport type {\n GenerateContentResponseUsageMetadata,\n ModalityTokenCount,\n} from '@google/genai'\n\n/**\n * Flattened modality token counts for normalized usage reporting.\n * Maps Gemini's ModalityTokenCount array to individual fields.\n */\nexport interface FlattenedModalityTokens {\n /** Text tokens */\n textTokens?: number\n /** Image tokens */\n imageTokens?: number\n /** Audio tokens */\n audioTokens?: number\n /** Video tokens */\n videoTokens?: number\n /** Document tokens (e.g. PDF inputs) */\n documentTokens?: number\n}\n\n/**\n * Flattens Gemini's ModalityTokenCount array into individual token fields.\n * Extracts TEXT, IMAGE, AUDIO, VIDEO, DOCUMENT modality counts into a\n * normalized structure.\n */\nexport function flattenModalityTokenCounts(\n modalities?: Array<ModalityTokenCount>,\n): FlattenedModalityTokens {\n if (!modalities || modalities.length === 0) {\n return {}\n }\n\n const result: FlattenedModalityTokens = {}\n\n for (const item of modalities) {\n if (!item.modality || item.tokenCount === undefined) {\n continue\n }\n\n const modality = item.modality.toUpperCase()\n const count = item.tokenCount\n\n switch (modality) {\n case 'TEXT':\n result.textTokens = (result.textTokens ?? 0) + count\n break\n case 'IMAGE':\n result.imageTokens = (result.imageTokens ?? 0) + count\n break\n case 'AUDIO':\n result.audioTokens = (result.audioTokens ?? 0) + count\n break\n case 'VIDEO':\n result.videoTokens = (result.videoTokens ?? 0) + count\n break\n case 'DOCUMENT':\n result.documentTokens = (result.documentTokens ?? 0) + count\n break\n }\n }\n\n return result\n}\n\n/**\n * Checks if a FlattenedModalityTokens object has any values set.\n */\nexport function hasModalityTokens(tokens: FlattenedModalityTokens): boolean {\n return (\n tokens.textTokens !== undefined ||\n tokens.imageTokens !== undefined ||\n tokens.audioTokens !== undefined ||\n tokens.videoTokens !== undefined ||\n tokens.documentTokens !== undefined\n )\n}\n\n/**\n * Gemini-specific provider usage details.\n * These fields are unique to Gemini and placed in providerUsageDetails.\n */\nexport type GeminiProviderUsageDetails = {\n /**\n * The traffic type for this request.\n * Can indicate whether request was handled by different service tiers.\n */\n trafficType?: string\n /**\n * Number of tokens in the results from tool executions,\n * which are provided back to the model as input.\n */\n toolUsePromptTokenCount?: number\n /**\n * Detailed breakdown by modality of the token counts from\n * the results of tool executions.\n */\n toolUsePromptTokensDetails?: Array<{\n modality: string\n tokenCount: number\n }>\n /**\n * Detailed breakdown of cache tokens by modality.\n * More granular than the normalized cachedTokens field.\n */\n cacheTokensDetails?: Array<{\n modality: string\n tokenCount: number\n }>\n}\n\n/**\n * Build normalized TokenUsage from Gemini's usageMetadata.\n * Handles modality breakdowns and thinking tokens. Returns `undefined` when the\n * provider reported no usage metadata, so callers omit the field rather than\n * fabricating zeroed totals.\n */\nexport function buildGeminiUsage(\n usageMetadata: GenerateContentResponseUsageMetadata | undefined | null,\n): TokenUsage<GeminiProviderUsageDetails> | undefined {\n if (!usageMetadata) return undefined\n\n const promptTokens = usageMetadata.promptTokenCount ?? 0\n const completionTokens = usageMetadata.candidatesTokenCount ?? 0\n\n const result = buildBaseUsage<GeminiProviderUsageDetails>({\n promptTokens: promptTokens,\n completionTokens: completionTokens,\n totalTokens:\n usageMetadata.totalTokenCount ?? promptTokens + completionTokens,\n })\n\n // Add prompt token details\n // Flatten modality breakdown for prompt\n const promptModalities = flattenModalityTokenCounts(\n usageMetadata.promptTokensDetails,\n )\n const cachedTokens = usageMetadata.cachedContentTokenCount\n\n const promptTokensDetails = {\n ...(hasModalityTokens(promptModalities) ? promptModalities : {}),\n ...(cachedTokens !== undefined && cachedTokens > 0 ? { cachedTokens } : {}),\n }\n\n // Add completion token details\n // Flatten modality breakdown for candidates (output)\n const completionModalities = flattenModalityTokenCounts(\n usageMetadata.candidatesTokensDetails,\n )\n const thoughtsTokens = usageMetadata.thoughtsTokenCount\n\n const completionTokensDetails = {\n ...(hasModalityTokens(completionModalities) ? completionModalities : {}),\n // Map thoughtsTokenCount to reasoningTokens for consistency with OpenAI\n ...(thoughtsTokens !== undefined && thoughtsTokens > 0\n ? { reasoningTokens: thoughtsTokens }\n : {}),\n }\n\n // Add provider-specific details\n const providerDetails: GeminiProviderUsageDetails = {\n ...(usageMetadata.trafficType\n ? { trafficType: usageMetadata.trafficType }\n : {}),\n ...(usageMetadata.toolUsePromptTokenCount !== undefined &&\n usageMetadata.toolUsePromptTokenCount > 0\n ? { toolUsePromptTokenCount: usageMetadata.toolUsePromptTokenCount }\n : {}),\n ...(usageMetadata.toolUsePromptTokensDetails &&\n usageMetadata.toolUsePromptTokensDetails.length > 0\n ? {\n toolUsePromptTokensDetails:\n usageMetadata.toolUsePromptTokensDetails.map((item) => ({\n modality: item.modality || 'UNKNOWN',\n tokenCount: item.tokenCount ?? 0,\n })),\n }\n : {}),\n ...(usageMetadata.cacheTokensDetails &&\n usageMetadata.cacheTokensDetails.length > 0\n ? {\n cacheTokensDetails: usageMetadata.cacheTokensDetails.map((item) => ({\n modality: item.modality || 'UNKNOWN',\n tokenCount: item.tokenCount ?? 0,\n })),\n }\n : {}),\n }\n\n // Add prompt token details if available\n if (Object.keys(promptTokensDetails).length > 0) {\n result.promptTokensDetails = promptTokensDetails\n }\n // Add provider details if available\n if (Object.keys(providerDetails).length > 0) {\n result.providerUsageDetails = providerDetails\n }\n // Add completion token details if available\n if (Object.keys(completionTokensDetails).length > 0) {\n result.completionTokensDetails = completionTokensDetails\n }\n\n return result\n}\n"],"mappings":";;;;;;;AA6BA,SAAgB,2BACd,YACyB;CACzB,IAAI,CAAC,cAAc,WAAW,WAAW,GACvC,OAAO,CAAC;CAGV,MAAM,SAAkC,CAAC;CAEzC,KAAK,MAAM,QAAQ,YAAY;EAC7B,IAAI,CAAC,KAAK,YAAY,KAAK,eAAe,KAAA,GACxC;EAGF,MAAM,WAAW,KAAK,SAAS,YAAY;EAC3C,MAAM,QAAQ,KAAK;EAEnB,QAAQ,UAAR;GACE,KAAK;IACH,OAAO,cAAc,OAAO,cAAc,KAAK;IAC/C;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK;IACH,OAAO,eAAe,OAAO,eAAe,KAAK;IACjD;GACF,KAAK,YACH,OAAO,kBAAkB,OAAO,kBAAkB,KAAK;EAE3D;CACF;CAEA,OAAO;AACT;;;;AAKA,SAAgB,kBAAkB,QAA0C;CAC1E,OACE,OAAO,eAAe,KAAA,KACtB,OAAO,gBAAgB,KAAA,KACvB,OAAO,gBAAgB,KAAA,KACvB,OAAO,gBAAgB,KAAA,KACvB,OAAO,mBAAmB,KAAA;AAE9B;;;;;;;AAyCA,SAAgB,iBACd,eACoD;CACpD,IAAI,CAAC,eAAe,OAAO,KAAA;CAE3B,MAAM,eAAe,cAAc,oBAAoB;CACvD,MAAM,mBAAmB,cAAc,wBAAwB;CAE/D,MAAM,SAAS,eAA2C;EAC1C;EACI;EAClB,aACE,cAAc,mBAAmB,eAAe;CACpD,CAAC;CAID,MAAM,mBAAmB,2BACvB,cAAc,mBAChB;CACA,MAAM,eAAe,cAAc;CAEnC,MAAM,sBAAsB;EAC1B,GAAI,kBAAkB,gBAAgB,IAAI,mBAAmB,CAAC;EAC9D,GAAI,iBAAiB,KAAA,KAAa,eAAe,IAAI,EAAE,aAAa,IAAI,CAAC;CAC3E;CAIA,MAAM,uBAAuB,2BAC3B,cAAc,uBAChB;CACA,MAAM,iBAAiB,cAAc;CAErC,MAAM,0BAA0B;EAC9B,GAAI,kBAAkB,oBAAoB,IAAI,uBAAuB,CAAC;EAEtE,GAAI,mBAAmB,KAAA,KAAa,iBAAiB,IACjD,EAAE,iBAAiB,eAAe,IAClC,CAAC;CACP;CAGA,MAAM,kBAA8C;EAClD,GAAI,cAAc,cACd,EAAE,aAAa,cAAc,YAAY,IACzC,CAAC;EACL,GAAI,cAAc,4BAA4B,KAAA,KAC9C,cAAc,0BAA0B,IACpC,EAAE,yBAAyB,cAAc,wBAAwB,IACjE,CAAC;EACL,GAAI,cAAc,8BAClB,cAAc,2BAA2B,SAAS,IAC9C,EACE,4BACE,cAAc,2BAA2B,KAAK,UAAU;GACtD,UAAU,KAAK,YAAY;GAC3B,YAAY,KAAK,cAAc;EACjC,EAAE,EACN,IACA,CAAC;EACL,GAAI,cAAc,sBAClB,cAAc,mBAAmB,SAAS,IACtC,EACE,oBAAoB,cAAc,mBAAmB,KAAK,UAAU;GAClE,UAAU,KAAK,YAAY;GAC3B,YAAY,KAAK,cAAc;EACjC,EAAE,EACJ,IACA,CAAC;CACP;CAGA,IAAI,OAAO,KAAK,mBAAmB,CAAC,CAAC,SAAS,GAC5C,OAAO,sBAAsB;CAG/B,IAAI,OAAO,KAAK,eAAe,CAAC,CAAC,SAAS,GACxC,OAAO,uBAAuB;CAGhC,IAAI,OAAO,KAAK,uBAAuB,CAAC,CAAC,SAAS,GAChD,OAAO,0BAA0B;CAGnC,OAAO;AACT"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai-gemini",
3
- "version": "0.23.0",
3
+ "version": "0.24.0",
4
4
  "description": "Google Gemini adapter for TanStack AI chat, images, speech, audio generation, and structured outputs.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -59,13 +59,13 @@
59
59
  "@tanstack/ai-utils": "^0.4.0"
60
60
  },
61
61
  "peerDependencies": {
62
- "@tanstack/ai": "^0.44.1"
62
+ "@tanstack/ai": "^0.45.0"
63
63
  },
64
64
  "devDependencies": {
65
- "@vitest/coverage-v8": "4.0.14",
66
- "vite": "^8.1.4",
65
+ "@vitest/coverage-v8": "4.1.10",
66
+ "vite": "^8.2.1",
67
67
  "zod": "^4.2.0",
68
- "@tanstack/ai": "0.44.1"
68
+ "@tanstack/ai": "0.45.0"
69
69
  },
70
70
  "scripts": {
71
71
  "build": "vite build",
@@ -7,18 +7,20 @@ import {
7
7
  } from '../utils'
8
8
  import { buildGeminiUsage } from '../usage'
9
9
  import {
10
+ isGeminiNativeImageModel,
10
11
  parseNativeImageSize,
11
12
  sizeToAspectRatio,
12
13
  validateImageSize,
13
14
  validateNumberOfImages,
14
15
  validatePrompt,
15
16
  } from '../image/image-provider-options'
16
- import type { GEMINI_IMAGE_MODELS } from '../model-meta'
17
+ import type { GeminiImageModels } from '../model-meta'
17
18
  import type {
19
+ GeminiAnyImageProviderOptions,
18
20
  GeminiImageModelInputModalitiesByName,
19
21
  GeminiImageModelProviderOptionsByName,
20
22
  GeminiImageModelSizeByName,
21
- GeminiImageProviderOptions,
23
+ GeminiNativeImageProviderOptions,
22
24
  } from '../image/image-provider-options'
23
25
  import type {
24
26
  GeneratedImage,
@@ -35,6 +37,7 @@ import type {
35
37
  GenerateImagesConfig,
36
38
  GenerateImagesResponse,
37
39
  GoogleGenAI,
40
+ ImageConfig,
38
41
  Part,
39
42
  } from '@google/genai'
40
43
  import type { GeminiClientConfig } from '../utils/client'
@@ -45,7 +48,7 @@ import type { GeminiClientConfig } from '../utils/client'
45
48
  export interface GeminiImageConfig extends GeminiClientConfig {}
46
49
 
47
50
  /** Model type for Gemini Image */
48
- export type GeminiImageModel = (typeof GEMINI_IMAGE_MODELS)[number]
51
+ export type GeminiImageModel = GeminiImageModels
49
52
 
50
53
  /**
51
54
  * Gemini Image Generation Adapter
@@ -65,7 +68,7 @@ export class GeminiImageAdapter<
65
68
  TModel extends GeminiImageModel,
66
69
  > extends BaseImageAdapter<
67
70
  TModel,
68
- GeminiImageProviderOptions,
71
+ GeminiAnyImageProviderOptions,
69
72
  GeminiImageModelProviderOptionsByName,
70
73
  GeminiImageModelSizeByName,
71
74
  GeminiImageModelInputModalitiesByName
@@ -75,7 +78,7 @@ export class GeminiImageAdapter<
75
78
 
76
79
  // Type-only property - never assigned at runtime
77
80
  declare '~types': {
78
- providerOptions: GeminiImageProviderOptions
81
+ providerOptions: GeminiAnyImageProviderOptions
79
82
  modelProviderOptionsByName: GeminiImageModelProviderOptionsByName
80
83
  modelSizeByName: GeminiImageModelSizeByName
81
84
  modelInputModalitiesByName: GeminiImageModelInputModalitiesByName
@@ -89,7 +92,7 @@ export class GeminiImageAdapter<
89
92
  }
90
93
 
91
94
  async generateImages(
92
- options: ImageGenerationOptions<GeminiImageProviderOptions>,
95
+ options: ImageGenerationOptions<GeminiAnyImageProviderOptions>,
93
96
  ): Promise<ImageGenerationResult> {
94
97
  const { model, logger } = options
95
98
 
@@ -121,7 +124,7 @@ export class GeminiImageAdapter<
121
124
  )
122
125
  }
123
126
 
124
- if (this.isGeminiImageModel(model)) {
127
+ if (isGeminiNativeImageModel(model)) {
125
128
  return await this.generateWithGeminiApi(options, resolved)
126
129
  }
127
130
 
@@ -155,29 +158,40 @@ export class GeminiImageAdapter<
155
158
  }
156
159
  }
157
160
 
158
- private isGeminiImageModel(model: string): boolean {
159
- return model.startsWith('gemini-')
160
- }
161
-
162
161
  private async generateWithGeminiApi(
163
- options: ImageGenerationOptions<GeminiImageProviderOptions>,
162
+ options: ImageGenerationOptions<GeminiNativeImageProviderOptions>,
164
163
  resolved: ResolvedMediaPrompt,
165
164
  ): Promise<ImageGenerationResult> {
166
165
  const { model, size, numberOfImages, modelOptions } = options
167
166
 
168
167
  const parsedSize = size ? parseNativeImageSize(size) : undefined
169
168
 
170
- // GeminiImageProviderOptions is Imagen-shaped — most fields
171
- // (personGeneration, safetyFilterLevel, addWatermark, outputMimeType,
172
- // outputCompressionQuality, guidanceScale, enhancePrompt,
173
- // includeSafetyAttributes, includeRaiReason, outputGcsUri, labels,
174
- // negativePrompt, language) are only valid on GenerateImagesConfig and
175
- // would be rejected by the Gemini-native generateContent path. Pick only
176
- // the fields that are valid on GenerateContentConfig instead of spreading
177
- // the whole options object.
178
- const nativeConfig: GenerateContentConfig = {}
179
- if (modelOptions?.seed !== undefined) {
180
- nativeConfig.seed = modelOptions.seed
169
+ // The portable `size` option is the baseline; modelOptions.imageConfig is
170
+ // the provider escape hatch and wins per field, so a caller passing only
171
+ // `imageConfig.imageSize` keeps the aspectRatio derived from `size`.
172
+ const imageConfig: ImageConfig = {
173
+ ...(parsedSize?.aspectRatio && { aspectRatio: parsedSize.aspectRatio }),
174
+ ...(parsedSize?.resolution && { imageSize: parsedSize.resolution }),
175
+ ...modelOptions?.imageConfig,
176
+ }
177
+
178
+ // Named picks, never a wholesale spread: the Imagen-shaped fields of
179
+ // GeminiImageProviderOptions (personGeneration, safetyFilterLevel,
180
+ // addWatermark, outputMimeType, …) are only valid on GenerateImagesConfig
181
+ // and would be rejected by generateContent. Picking by name means no
182
+ // Imagen field can reach this path even if one slips past the per-model
183
+ // provider-options map.
184
+ const nativeConfig: GenerateContentConfig = {
185
+ ...(modelOptions?.seed !== undefined && { seed: modelOptions.seed }),
186
+ ...(modelOptions?.safetySettings !== undefined && {
187
+ safetySettings: modelOptions.safetySettings,
188
+ }),
189
+ ...(modelOptions?.thinkingConfig !== undefined && {
190
+ thinkingConfig: modelOptions.thinkingConfig,
191
+ }),
192
+ ...(modelOptions?.systemInstruction !== undefined && {
193
+ systemInstruction: modelOptions.systemInstruction,
194
+ }),
181
195
  }
182
196
 
183
197
  const config: GenerateContentConfig = {
@@ -186,16 +200,7 @@ export class GeminiImageAdapter<
186
200
  // IMPORTANT: responseModalities is a protected default — set it AFTER
187
201
  // nativeConfig so nothing can silently disable image output.
188
202
  responseModalities: ['TEXT', 'IMAGE'],
189
- ...(parsedSize && {
190
- imageConfig: {
191
- ...(parsedSize.aspectRatio && {
192
- aspectRatio: parsedSize.aspectRatio,
193
- }),
194
- ...(parsedSize.resolution && {
195
- imageSize: parsedSize.resolution,
196
- }),
197
- },
198
- }),
203
+ ...(Object.keys(imageConfig).length > 0 && { imageConfig }),
199
204
  }
200
205
 
201
206
  const contents = this.buildContents(resolved, numberOfImages)
@@ -321,7 +326,7 @@ export class GeminiImageAdapter<
321
326
  }
322
327
 
323
328
  private buildImagenConfig(
324
- options: ImageGenerationOptions<GeminiImageProviderOptions>,
329
+ options: ImageGenerationOptions<GeminiAnyImageProviderOptions>,
325
330
  ): GenerateImagesConfig {
326
331
  const { size, numberOfImages, modelOptions } = options
327
332
 
@@ -329,11 +334,62 @@ export class GeminiImageAdapter<
329
334
  // vendor `GenerateImagesConfig` fields are `field?: T` (no `| undefined`),
330
335
  // so we can only assign the property when we actually have a value.
331
336
  const sizeAspectRatio = size ? sizeToAspectRatio(size) : undefined
337
+
338
+ // Named picks, never a wholesale spread — the mirror image of the native
339
+ // path below. A native-only field (safetySettings, thinkingConfig,
340
+ // imageConfig, systemInstruction) belongs to GenerateContentConfig and is
341
+ // rejected by generateImages with 400 INVALID_ARGUMENT, so it must not be
342
+ // able to reach here even when the caller's `modelOptions` was typed
343
+ // against both shapes at once (e.g. an adapter inferred from a union of
344
+ // model names).
332
345
  return {
333
346
  numberOfImages: numberOfImages ?? 1,
334
- // Map size to aspect ratio if provided (modelOptions.aspectRatio will override)
347
+ // Map size to aspect ratio if provided; modelOptions.aspectRatio,
348
+ // picked after it, overrides.
335
349
  ...(sizeAspectRatio !== undefined && { aspectRatio: sizeAspectRatio }),
336
- ...modelOptions,
350
+ ...(modelOptions?.aspectRatio !== undefined && {
351
+ aspectRatio: modelOptions.aspectRatio,
352
+ }),
353
+ ...(modelOptions?.personGeneration !== undefined && {
354
+ personGeneration: modelOptions.personGeneration,
355
+ }),
356
+ ...(modelOptions?.safetyFilterLevel !== undefined && {
357
+ safetyFilterLevel: modelOptions.safetyFilterLevel,
358
+ }),
359
+ ...(modelOptions?.seed !== undefined && { seed: modelOptions.seed }),
360
+ ...(modelOptions?.addWatermark !== undefined && {
361
+ addWatermark: modelOptions.addWatermark,
362
+ }),
363
+ ...(modelOptions?.language !== undefined && {
364
+ language: modelOptions.language,
365
+ }),
366
+ ...(modelOptions?.negativePrompt !== undefined && {
367
+ negativePrompt: modelOptions.negativePrompt,
368
+ }),
369
+ ...(modelOptions?.outputMimeType !== undefined && {
370
+ outputMimeType: modelOptions.outputMimeType,
371
+ }),
372
+ ...(modelOptions?.outputCompressionQuality !== undefined && {
373
+ outputCompressionQuality: modelOptions.outputCompressionQuality,
374
+ }),
375
+ ...(modelOptions?.guidanceScale !== undefined && {
376
+ guidanceScale: modelOptions.guidanceScale,
377
+ }),
378
+ ...(modelOptions?.enhancePrompt !== undefined && {
379
+ enhancePrompt: modelOptions.enhancePrompt,
380
+ }),
381
+ ...(modelOptions?.includeSafetyAttributes !== undefined && {
382
+ includeSafetyAttributes: modelOptions.includeSafetyAttributes,
383
+ }),
384
+ ...(modelOptions?.includeRaiReason !== undefined && {
385
+ includeRaiReason: modelOptions.includeRaiReason,
386
+ }),
387
+ ...(modelOptions?.outputGcsUri !== undefined && {
388
+ outputGcsUri: modelOptions.outputGcsUri,
389
+ }),
390
+ ...(modelOptions?.labels !== undefined && {
391
+ labels: modelOptions.labels,
392
+ }),
337
393
  }
338
394
  }
339
395
 
@@ -392,6 +448,18 @@ export class GeminiImageAdapter<
392
448
  }
393
449
  }
394
450
 
451
+ /** @deprecated Shut down 2026-06-25. Use `gemini-3.1-flash-image`. */
452
+ export function createGeminiImage(
453
+ model: 'gemini-3.1-flash-image-preview',
454
+ apiKey: string,
455
+ config?: Omit<GeminiImageConfig, 'apiKey'>,
456
+ ): GeminiImageAdapter<'gemini-3.1-flash-image-preview'>
457
+ /** @deprecated Shut down 2026-06-25. Use `gemini-3-pro-image`. */
458
+ export function createGeminiImage(
459
+ model: 'gemini-3-pro-image-preview',
460
+ apiKey: string,
461
+ config?: Omit<GeminiImageConfig, 'apiKey'>,
462
+ ): GeminiImageAdapter<'gemini-3-pro-image-preview'>
395
463
  /**
396
464
  * Creates a Gemini image adapter with explicit API key.
397
465
  * Type resolution happens here at the call site.
@@ -411,6 +479,11 @@ export class GeminiImageAdapter<
411
479
  * });
412
480
  * ```
413
481
  */
482
+ export function createGeminiImage<TModel extends GeminiImageModel>(
483
+ model: TModel,
484
+ apiKey: string,
485
+ config?: Omit<GeminiImageConfig, 'apiKey'>,
486
+ ): GeminiImageAdapter<TModel>
414
487
  export function createGeminiImage<TModel extends GeminiImageModel>(
415
488
  model: TModel,
416
489
  apiKey: string,
@@ -419,6 +492,16 @@ export function createGeminiImage<TModel extends GeminiImageModel>(
419
492
  return new GeminiImageAdapter({ apiKey, ...config }, model)
420
493
  }
421
494
 
495
+ /** @deprecated Shut down 2026-06-25. Use `gemini-3.1-flash-image`. */
496
+ export function geminiImage(
497
+ model: 'gemini-3.1-flash-image-preview',
498
+ config?: Omit<GeminiImageConfig, 'apiKey'>,
499
+ ): GeminiImageAdapter<'gemini-3.1-flash-image-preview'>
500
+ /** @deprecated Shut down 2026-06-25. Use `gemini-3-pro-image`. */
501
+ export function geminiImage(
502
+ model: 'gemini-3-pro-image-preview',
503
+ config?: Omit<GeminiImageConfig, 'apiKey'>,
504
+ ): GeminiImageAdapter<'gemini-3-pro-image-preview'>
422
505
  /**
423
506
  * Creates a Gemini image adapter with automatic API key detection from environment variables.
424
507
  * Type resolution happens here at the call site.
@@ -443,6 +526,10 @@ export function createGeminiImage<TModel extends GeminiImageModel>(
443
526
  * });
444
527
  * ```
445
528
  */
529
+ export function geminiImage<TModel extends GeminiImageModel>(
530
+ model: TModel,
531
+ config?: Omit<GeminiImageConfig, 'apiKey'>,
532
+ ): GeminiImageAdapter<TModel>
446
533
  export function geminiImage<TModel extends GeminiImageModel>(
447
534
  model: TModel,
448
535
  config?: Omit<GeminiImageConfig, 'apiKey'>,
@@ -1,12 +1,24 @@
1
1
  import type { GeminiImageModels } from '../model-meta'
2
2
  import type {
3
+ ContentUnion,
4
+ ImageConfig,
3
5
  ImagePromptLanguage,
4
6
  PersonGeneration,
5
7
  SafetyFilterLevel,
8
+ SafetySetting,
9
+ ThinkingConfig,
6
10
  } from '@google/genai'
7
11
 
8
12
  // Re-export SDK types so users can use them directly
9
- export type { ImagePromptLanguage, PersonGeneration, SafetyFilterLevel }
13
+ export type {
14
+ ContentUnion,
15
+ ImageConfig,
16
+ ImagePromptLanguage,
17
+ PersonGeneration,
18
+ SafetyFilterLevel,
19
+ SafetySetting,
20
+ ThinkingConfig,
21
+ }
10
22
 
11
23
  /**
12
24
  * Gemini Imagen aspect ratio options
@@ -121,11 +133,80 @@ export interface GeminiImageProviderOptions {
121
133
  }
122
134
 
123
135
  /**
124
- * Model-specific provider options mapping
125
- * Currently all Imagen models use the same options structure
136
+ * Provider options for Gemini native image models (Nano Banana and friends).
137
+ *
138
+ * These models are served by `generateContent`, not `generateImages`, so they
139
+ * are configured by @google/genai's `GenerateContentConfig` — a different
140
+ * shape from the Imagen-only {@link GeminiImageProviderOptions} above. Only
141
+ * the `GenerateContentConfig` fields with clear image-generation semantics are
142
+ * surfaced; sampling knobs (`temperature`, `topK`, …) and chat-only plumbing
143
+ * (`tools`, `responseSchema`, …) are deliberately left out.
144
+ *
145
+ * `responseModalities` is intentionally absent: the adapter always requests
146
+ * `['TEXT', 'IMAGE']`, and letting a caller override it would silently disable
147
+ * image output on an image-generation call.
148
+ */
149
+ export interface GeminiNativeImageProviderOptions {
150
+ /**
151
+ * Optional seed for reproducible image generation
152
+ * When the same seed is used with the same prompt and settings,
153
+ * you should get similar (though not identical) results
154
+ */
155
+ seed?: number
156
+
157
+ /**
158
+ * Per-category safety thresholds applied to the request
159
+ * Each entry pairs a HarmCategory with a HarmBlockThreshold
160
+ */
161
+ safetySettings?: Array<SafetySetting>
162
+
163
+ /**
164
+ * Controls the model's internal reasoning before it emits an image
165
+ * Use to raise or disable the thinking budget on models that support it
166
+ */
167
+ thinkingConfig?: ThinkingConfig
168
+
169
+ /**
170
+ * Native image output controls. Merged over the values derived from the
171
+ * portable `size` option, so fields set here win per field while the rest
172
+ * of `size` is preserved.
173
+ *
174
+ * Only `aspectRatio` and `imageSize` are accepted on the Gemini Developer
175
+ * API. Other SDK `ImageConfig` keys throw on this surface.
176
+ */
177
+ imageConfig?: GeminiNativeImageConfig
178
+
179
+ /**
180
+ * System-level instructions that steer the model for the whole request,
181
+ * e.g. a house art direction applied on top of the per-call prompt
182
+ */
183
+ systemInstruction?: ContentUnion
184
+ }
185
+
186
+ /**
187
+ * Every provider-option field this adapter understands, across both API
188
+ * paths. Used as the adapter's base (model-agnostic) option type; the
189
+ * per-model map below is what narrows a given model to the half that
190
+ * actually applies to it.
191
+ */
192
+ export type GeminiAnyImageProviderOptions = GeminiImageProviderOptions &
193
+ GeminiNativeImageProviderOptions
194
+
195
+ /**
196
+ * Model-specific provider options mapping.
197
+ * Gemini native image models go through `generateContent` and take
198
+ * `GenerateContentConfig` fields; Imagen models go through `generateImages`
199
+ * and take `GenerateImagesConfig` fields. Mirrors the native/Imagen split in
200
+ * {@link GeminiImageModelSizeByName} and
201
+ * {@link GeminiImageModelInputModalitiesByName}.
126
202
  */
127
203
  export type GeminiImageModelProviderOptionsByName = {
128
- [K in GeminiImageModels]: GeminiImageProviderOptions
204
+ [K in GeminiNativeImageModels]: GeminiNativeImageProviderOptions
205
+ } & {
206
+ [K in Exclude<
207
+ GeminiImageModels,
208
+ GeminiNativeImageModels
209
+ >]: GeminiImageProviderOptions
129
210
  }
130
211
 
131
212
  /**
@@ -145,47 +226,157 @@ export type GeminiImageSize =
145
226
  | '1080x1920'
146
227
 
147
228
  /**
148
- * Aspect ratios supported by Gemini native image models (via generateContent API).
149
- * Matches the SDK's ImageConfig.aspectRatio values.
229
+ * The ten aspect ratios every Gemini native image model accepts.
230
+ *
231
+ * Note `9:21` is deliberately absent: it exists only on Vertex / Cloud and is
232
+ * rejected by the Gemini API (`generateContent`), which is the surface this
233
+ * adapter targets.
234
+ *
235
+ * @see https://ai.google.dev/gemini-api/docs/image-generation
150
236
  */
151
- export type GeminiNativeImageAspectRatio =
237
+ export type GeminiStandardImageAspectRatio =
152
238
  | '1:1'
153
239
  | '2:3'
154
240
  | '3:2'
155
241
  | '3:4'
156
242
  | '4:3'
243
+ | '4:5'
244
+ | '5:4'
157
245
  | '9:16'
158
246
  | '16:9'
159
247
  | '21:9'
160
248
 
161
249
  /**
162
- * Resolution tiers for Gemini native image models.
163
- * Matches the SDK's ImageConfig.imageSize values.
250
+ * The ten standard ratios plus the four extreme banner/strip ratios that only
251
+ * the Gemini 3.1 Flash Image models accept — 14 values, matching the
252
+ * `generateContent` `ImageConfig.aspectRatio` field union.
253
+ *
254
+ * @see https://ai.google.dev/api/generate-content
255
+ */
256
+ export type GeminiExtendedImageAspectRatio =
257
+ | GeminiStandardImageAspectRatio
258
+ | '1:4'
259
+ | '4:1'
260
+ | '1:8'
261
+ | '8:1'
262
+
263
+ /**
264
+ * Sizes for `gemini-3.1-flash-image` (and its shut-down `-preview` alias):
265
+ * all 14 aspect ratios at 512 / 1K / 2K / 4K. `512` is the wire token for the
266
+ * 0.5K tier — not `512px`, and the `K` is case-sensitive (`1k` is rejected).
267
+ */
268
+ export type Gemini31FlashImageSize =
269
+ `${GeminiExtendedImageAspectRatio}_${'512' | '1K' | '2K' | '4K'}`
270
+
271
+ /**
272
+ * Sizes for `gemini-3.1-flash-lite-image`: all 14 aspect ratios, 1K only.
273
+ * 2K and 4K are unsupported on this model.
274
+ *
275
+ * The four banner ratios (`1:4`, `4:1`, `1:8`, `8:1`) come from the Cloud
276
+ * model page. The Gemini API page states a count of 14 but does not list them.
277
+ *
278
+ * @see https://docs.cloud.google.com/gemini-enterprise-agent-platform/models/gemini/3-1-flash-lite-image
279
+ * @see https://ai.google.dev/gemini-api/docs/models/gemini-3.1-flash-lite-image
280
+ */
281
+ export type Gemini31FlashLiteImageSize = `${GeminiExtendedImageAspectRatio}_1K`
282
+
283
+ /**
284
+ * Sizes for `gemini-3-pro-image` (and its shut-down `-preview` alias): the ten
285
+ * standard aspect ratios at 1K / 2K / 4K. Pro has no 512 tier and none of the
286
+ * extreme banner ratios on the Gemini API.
287
+ */
288
+ export type Gemini3ProImageSize =
289
+ `${GeminiStandardImageAspectRatio}_${'1K' | '2K' | '4K'}`
290
+
291
+ /**
292
+ * Sizes for `gemini-2.5-flash-image`: a bare aspect ratio with no resolution
293
+ * suffix, e.g. `'16:9'`. Google documents no `image_size` value or default for
294
+ * this model — it emits a single fixed 1024px-class output — so the adapter
295
+ * sends `imageConfig.aspectRatio` and omits `imageSize` entirely rather than
296
+ * guessing a tier the API never documented.
164
297
  */
165
- export type GeminiNativeImageResolution = '1K' | '2K' | '4K'
298
+ export type Gemini25FlashImageSize = GeminiStandardImageAspectRatio
166
299
 
167
300
  /**
168
- * Template literal size type for Gemini native image models: "16:9_4K", "1:1_2K", etc.
301
+ * `imageConfig` fields the Gemini Developer API accepts on `generateContent`.
302
+ * Other `@google/genai` `ImageConfig` keys (`personGeneration`,
303
+ * `outputMimeType`, and more) throw on this surface.
304
+ */
305
+ export type GeminiNativeImageConfig = {
306
+ aspectRatio?: GeminiExtendedImageAspectRatio
307
+ imageSize?: '512' | '1K' | '2K' | '4K'
308
+ }
309
+
310
+ /**
311
+ * Any size accepted by any Gemini native image model. Prefer the per-model
312
+ * narrowing in {@link GeminiImageModelSizeByName} — this union is the widest
313
+ * possible set and accepts combinations no single model supports.
169
314
  */
170
315
  export type GeminiNativeImageSize =
171
- `${GeminiNativeImageAspectRatio}_${GeminiNativeImageResolution}`
316
+ | Gemini31FlashImageSize
317
+ | Gemini31FlashLiteImageSize
318
+ | Gemini3ProImageSize
319
+ | Gemini25FlashImageSize
172
320
 
173
321
  /**
174
322
  * Gemini native image models that use the generateContent API path.
175
- * These models support template literal sizes (aspectRatio_resolution).
323
+ * These models take an aspect-ratio-based size rather than Imagen's
324
+ * WIDTHxHEIGHT pixel strings.
325
+ *
326
+ * This array is the single source of truth for the native/Imagen split: the
327
+ * `GeminiNativeImageModels` union and the per-model option/size/modality maps
328
+ * all derive from it. The `satisfies` clause makes a typo (or a name that
329
+ * is not a known image model) a build error rather than a phantom key on every
330
+ * per-model map.
331
+ *
332
+ * It is also the single source of truth for the adapter's runtime routing
333
+ * — see {@link isGeminiNativeImageModel}. Adding a new `gemini-*` image model
334
+ * means adding it here as well as to `GEMINI_IMAGE_MODELS` in model-meta.
335
+ * Until it is listed here it routes to the Imagen API instead and fails
336
+ * loudly on the first call, rather than silently taking the wrong option
337
+ * shape.
176
338
  */
339
+ export const GEMINI_NATIVE_IMAGE_MODELS = [
340
+ 'gemini-3.1-flash-image',
341
+ 'gemini-3.1-flash-image-preview',
342
+ 'gemini-3.1-flash-lite-image',
343
+ 'gemini-3-pro-image',
344
+ 'gemini-3-pro-image-preview',
345
+ 'gemini-2.5-flash-image',
346
+ ] as const satisfies ReadonlyArray<GeminiImageModels>
347
+
177
348
  export type GeminiNativeImageModels =
178
- | 'gemini-3.1-flash-image-preview'
179
- | 'gemini-3.1-flash-lite-image'
180
- | 'gemini-3-pro-image-preview'
181
- | 'gemini-2.5-flash-image'
349
+ (typeof GEMINI_NATIVE_IMAGE_MODELS)[number]
350
+
351
+ const NATIVE_IMAGE_MODEL_NAMES: ReadonlySet<string> = new Set(
352
+ GEMINI_NATIVE_IMAGE_MODELS,
353
+ )
182
354
 
183
355
  /**
184
- * Model-specific size options mapping.
185
- * Gemini native image models use template literal sizes, Imagen models use pixel sizes.
356
+ * Runtime counterpart to {@link GeminiNativeImageModels} — decides which of
357
+ * the two Gemini image APIs a model goes to.
358
+ *
359
+ * Membership in {@link GEMINI_NATIVE_IMAGE_MODELS}, not a `gemini-` prefix
360
+ * test, so the runtime route and the type-level split cannot drift apart. An
361
+ * id this package does not know about reaches the Imagen endpoint and fails
362
+ * there, which is the intended signal to add the model here rather than to
363
+ * have it silently take the native path with Imagen-shaped option types.
364
+ */
365
+ export function isGeminiNativeImageModel(model: string): boolean {
366
+ return NATIVE_IMAGE_MODEL_NAMES.has(model)
367
+ }
368
+
369
+ /**
370
+ * Model-specific size options mapping. Each native model gets its own ratio ×
371
+ * resolution set (they genuinely differ); Imagen models use pixel sizes.
186
372
  */
187
373
  export type GeminiImageModelSizeByName = {
188
- [K in GeminiNativeImageModels]: GeminiNativeImageSize
374
+ 'gemini-3.1-flash-image': Gemini31FlashImageSize
375
+ 'gemini-3.1-flash-image-preview': Gemini31FlashImageSize
376
+ 'gemini-3.1-flash-lite-image': Gemini31FlashLiteImageSize
377
+ 'gemini-3-pro-image': Gemini3ProImageSize
378
+ 'gemini-3-pro-image-preview': Gemini3ProImageSize
379
+ 'gemini-2.5-flash-image': Gemini25FlashImageSize
189
380
  } & {
190
381
  [K in Exclude<GeminiImageModels, GeminiNativeImageModels>]: GeminiImageSize
191
382
  }
@@ -309,13 +500,23 @@ export function validatePrompt(options: {
309
500
 
310
501
  /**
311
502
  * Parses a Gemini native image size string into its components.
312
- * Format: "aspectRatio_resolution" e.g. "16:9_4K" → { aspectRatio: "16:9", resolution: "4K" }
503
+ *
504
+ * Format: `"aspectRatio_resolution"`, e.g. `"16:9_4K"` →
505
+ * `{ aspectRatio: "16:9", resolution: "4K" }`.
506
+ *
507
+ * The resolution suffix is optional: `gemini-2.5-flash-image` takes a bare
508
+ * aspect ratio (`"16:9"` → `{ aspectRatio: "16:9" }`) because Google documents
509
+ * no `image_size` for it, and the caller must then omit `imageSize` from the
510
+ * request rather than substituting a default.
313
511
  */
314
512
  export function parseNativeImageSize(
315
513
  size: string,
316
- ): { aspectRatio: string; resolution: string } | undefined {
317
- const match = size.match(/^(\d+:\d+)_(.+)$/)
514
+ ): { aspectRatio: string; resolution?: string } | undefined {
515
+ const match = size.match(/^(\d+:\d+)(?:_(.+))?$/)
318
516
  const [, aspectRatio, resolution] = match ?? []
319
- if (aspectRatio === undefined || resolution === undefined) return undefined
320
- return { aspectRatio, resolution }
517
+ if (aspectRatio === undefined) return undefined
518
+ return {
519
+ aspectRatio,
520
+ ...(resolution !== undefined && { resolution }),
521
+ }
321
522
  }