@ai-sdk/google 3.0.118 → 3.0.120

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1111,8 +1111,8 @@ The [Gemini Interactions API](https://ai.google.dev/gemini-api/docs/interactions
1111
1111
  (`POST /v1beta/interactions`) is a separate Google endpoint with server-side
1112
1112
  state, unified content blocks, first-class built-in tools, agent presets,
1113
1113
  managed agents that run in a sandboxed Linux environment, and native
1114
- multimodal image output. It is reached via the `google.interactions(...)`
1115
- factory:
1114
+ multimodal image and video output. It is reached via the
1115
+ `google.interactions(...)` factory:
1116
1116
 
1117
1117
  ```ts
1118
1118
  import { google } from '@ai-sdk/google';
@@ -1202,16 +1202,42 @@ The following optional provider options are available:
1202
1202
  Whether the model returns synthesized thought summaries on reasoning
1203
1203
  parts. Defaults to the API default.
1204
1204
 
1205
- - **responseFormat** _Array\<\{ type: 'text' | 'image' | 'audio'; mimeType?: string; schema?: unknown; aspectRatio?: string; imageSize?: '1K' \| '2K' \| '4K' \| '512' \}\>_
1205
+ - **responseFormat** _Array\<ResponseFormatEntry\>_
1206
1206
 
1207
1207
  Output-format entries that map directly to the API's `response_format`
1208
- array. Use this for fine-grained control over image, audio, or non-JSON
1209
- text outputs (e.g. `aspectRatio` and `imageSize` for image generation).
1208
+ array. Entries have the following shape:
1209
+
1210
+ ```ts
1211
+ type ResponseFormatEntry =
1212
+ | { type: 'text'; mimeType?: string; schema?: unknown }
1213
+ | {
1214
+ type: 'image';
1215
+ mimeType?: string;
1216
+ aspectRatio?: string;
1217
+ imageSize?: '1K' | '2K' | '4K' | '512';
1218
+ }
1219
+ | { type: 'audio'; mimeType?: string }
1220
+ | {
1221
+ type: 'video';
1222
+ aspectRatio?: '16:9' | '9:16';
1223
+ resolution?: '360p' | '720p' | '1080p' | '4k';
1224
+ duration?: string;
1225
+ delivery?: 'inline' | 'uri';
1226
+ gcsUri?: string;
1227
+ };
1228
+ ```
1229
+
1230
+ Use this for fine-grained control over image, audio, video, or non-JSON
1231
+ text outputs. Video entries can control aspect ratio, resolution, duration,
1232
+ and inline or URI delivery; set `gcsUri` when delivering to Google Cloud
1233
+ Storage. Inline videos are returned in `result.files` by `generateText` and
1234
+ emitted as `file` parts by `streamText`.
1210
1235
  The AI SDK call-level `responseFormat: { type: 'json', schema }` still
1211
1236
  drives JSON-mode automatically and prepends a matching text entry;
1212
1237
  entries listed here are appended.
1213
1238
 
1214
- `aspectRatio` accepts `1:1`, `2:3`, `3:2`, `3:4`, `4:3`, `4:5`, `5:4`,
1239
+ For image entries, `aspectRatio` accepts `1:1`, `2:3`, `3:2`, `3:4`, `4:3`,
1240
+ `4:5`, `5:4`,
1215
1241
  `9:16`, `16:9`, `21:9`, `1:8`, `8:1`, `1:4`, `4:1`.
1216
1242
 
1217
1243
  - **imageConfig** _\{ aspectRatio?: string; imageSize?: '1K' | '2K' | '4K' | '512' \}_ (deprecated)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/google",
3
- "version": "3.0.118",
3
+ "version": "3.0.120",
4
4
  "license": "Apache-2.0",
5
5
  "sideEffects": false,
6
6
  "main": "./dist/index.js",
@@ -37,7 +37,7 @@
37
37
  },
38
38
  "dependencies": {
39
39
  "@ai-sdk/provider": "3.0.15",
40
- "@ai-sdk/provider-utils": "4.0.49"
40
+ "@ai-sdk/provider-utils": "4.0.50"
41
41
  },
42
42
  "devDependencies": {
43
43
  "@types/node": "20.17.24",
@@ -8,13 +8,16 @@ export type GoogleGenerativeAITokenDetail = {
8
8
  export type GoogleGenerativeAIUsageMetadata = {
9
9
  promptTokenCount?: number | null;
10
10
  candidatesTokenCount?: number | null;
11
+ toolUsePromptTokenCount?: number | null;
11
12
  totalTokenCount?: number | null;
12
13
  cachedContentTokenCount?: number | null;
13
14
  thoughtsTokenCount?: number | null;
14
15
  trafficType?: string | null;
15
16
  serviceTier?: string | null;
16
17
  promptTokensDetails?: GoogleGenerativeAITokenDetail[] | null;
18
+ cacheTokensDetails?: GoogleGenerativeAITokenDetail[] | null;
17
19
  candidatesTokensDetails?: GoogleGenerativeAITokenDetail[] | null;
20
+ toolUsePromptTokensDetails?: GoogleGenerativeAITokenDetail[] | null;
18
21
  };
19
22
 
20
23
  export function convertGoogleGenerativeAIUsage(
@@ -1481,26 +1481,33 @@ const getSafetyRatingSchema = () =>
1481
1481
 
1482
1482
  const tokenDetailsSchema = z
1483
1483
  .array(
1484
- z.object({
1485
- modality: z.string(),
1486
- tokenCount: z.number(),
1487
- }),
1484
+ z
1485
+ .object({
1486
+ modality: z.string(),
1487
+ tokenCount: z.number(),
1488
+ })
1489
+ .catchall(z.json()),
1488
1490
  )
1489
1491
  .nullish();
1490
1492
 
1491
- const usageSchema = z.object({
1492
- cachedContentTokenCount: z.number().nullish(),
1493
- thoughtsTokenCount: z.number().nullish(),
1494
- promptTokenCount: z.number().nullish(),
1495
- candidatesTokenCount: z.number().nullish(),
1496
- totalTokenCount: z.number().nullish(),
1497
- // https://cloud.google.com/vertex-ai/generative-ai/docs/reference/rest/v1/GenerateContentResponse#TrafficType
1498
- trafficType: z.string().nullish(),
1499
- serviceTier: z.string().nullish(),
1500
- // https://ai.google.dev/api/generate-content#Modality
1501
- promptTokensDetails: tokenDetailsSchema,
1502
- candidatesTokensDetails: tokenDetailsSchema,
1503
- });
1493
+ const usageSchema = z
1494
+ .object({
1495
+ cachedContentTokenCount: z.number().nullish(),
1496
+ thoughtsTokenCount: z.number().nullish(),
1497
+ promptTokenCount: z.number().nullish(),
1498
+ candidatesTokenCount: z.number().nullish(),
1499
+ toolUsePromptTokenCount: z.number().nullish(),
1500
+ totalTokenCount: z.number().nullish(),
1501
+ // https://cloud.google.com/vertex-ai/generative-ai/docs/reference/rest/v1/GenerateContentResponse#TrafficType
1502
+ trafficType: z.string().nullish(),
1503
+ serviceTier: z.string().nullish(),
1504
+ // https://ai.google.dev/api/generate-content#Modality
1505
+ promptTokensDetails: tokenDetailsSchema,
1506
+ cacheTokensDetails: tokenDetailsSchema,
1507
+ candidatesTokensDetails: tokenDetailsSchema,
1508
+ toolUsePromptTokensDetails: tokenDetailsSchema,
1509
+ })
1510
+ .catchall(z.json());
1504
1511
 
1505
1512
  // https://ai.google.dev/api/generate-content#UrlRetrievalMetadata
1506
1513
  export const getUrlContextMetadataSchema = () =>
@@ -75,8 +75,8 @@ export const googleInteractionsLanguageModelOptions = lazySchema(() =>
75
75
 
76
76
  /**
77
77
  * Output-format entries that map directly to the API's `response_format`
78
- * array. Use this to request image, audio, or non-JSON text outputs
79
- * with full control over `mime_type`, `aspect_ratio`, and `image_size`.
78
+ * array. Use this to request image, audio, video, or non-JSON text
79
+ * outputs with modality-specific controls.
80
80
  *
81
81
  * Entries are sent in order. The AI SDK call-level `responseFormat: {
82
82
  * type: 'json', schema }` still drives JSON-mode and adds a matching
@@ -123,6 +123,16 @@ export const googleInteractionsLanguageModelOptions = lazySchema(() =>
123
123
  mimeType: z.string().nullish(),
124
124
  })
125
125
  .loose(),
126
+ z
127
+ .object({
128
+ type: z.literal('video'),
129
+ aspectRatio: z.enum(['16:9', '9:16']).nullish(),
130
+ resolution: z.enum(['360p', '720p', '1080p', '4k']).nullish(),
131
+ duration: z.string().nullish(),
132
+ delivery: z.enum(['inline', 'uri']).nullish(),
133
+ gcsUri: z.string().nullish(),
134
+ })
135
+ .loose(),
126
136
  ]),
127
137
  )
128
138
  .nullish(),
@@ -221,6 +221,17 @@ export class GoogleInteractionsLanguageModel implements LanguageModelV3 {
221
221
  mime_type: entry.mimeType ?? undefined,
222
222
  }),
223
223
  );
224
+ } else if (entry.type === 'video') {
225
+ responseFormatEntries.push(
226
+ pruneUndefined({
227
+ type: 'video' as const,
228
+ aspect_ratio: entry.aspectRatio ?? undefined,
229
+ resolution: entry.resolution ?? undefined,
230
+ duration: entry.duration ?? undefined,
231
+ delivery: entry.delivery ?? undefined,
232
+ gcs_uri: entry.gcsUri ?? undefined,
233
+ }),
234
+ );
224
235
  }
225
236
  }
226
237
  }
@@ -434,6 +434,10 @@ export type GoogleInteractionsImageSize = '1K' | '2K' | '4K' | '512';
434
434
  *
435
435
  * { type: 'image', mime_type, aspect_ratio?, image_size? }
436
436
  * -- image generation. `mime_type` defaults to `image/png`.
437
+ *
438
+ * { type: 'video', aspect_ratio?, resolution?, duration?, delivery?,
439
+ * gcs_uri? }
440
+ * -- video generation and delivery configuration.
437
441
  */
438
442
  export type GoogleInteractionsResponseFormatTextEntry = {
439
443
  type: 'text';
@@ -453,10 +457,20 @@ export type GoogleInteractionsResponseFormatAudioEntry = {
453
457
  mime_type?: string;
454
458
  };
455
459
 
460
+ export type GoogleInteractionsResponseFormatVideoEntry = {
461
+ type: 'video';
462
+ aspect_ratio?: '16:9' | '9:16';
463
+ resolution?: '360p' | '720p' | '1080p' | '4k';
464
+ duration?: string;
465
+ delivery?: 'inline' | 'uri';
466
+ gcs_uri?: string;
467
+ };
468
+
456
469
  export type GoogleInteractionsResponseFormatEntry =
457
470
  | GoogleInteractionsResponseFormatTextEntry
458
471
  | GoogleInteractionsResponseFormatImageEntry
459
- | GoogleInteractionsResponseFormatAudioEntry;
472
+ | GoogleInteractionsResponseFormatAudioEntry
473
+ | GoogleInteractionsResponseFormatVideoEntry;
460
474
 
461
475
  export type GoogleInteractionsGenerationConfig = {
462
476
  temperature?: number;