@ai-sdk/google-vertex 5.0.97 → 5.0.99

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/dist/anthropic/edge/index.d.ts +84 -80
  3. package/dist/anthropic/edge/index.d.ts.map +1 -0
  4. package/dist/anthropic/edge/index.js +232 -272
  5. package/dist/anthropic/edge/index.js.map +1 -1
  6. package/dist/anthropic/index.d.ts +69 -66
  7. package/dist/anthropic/index.d.ts.map +1 -0
  8. package/dist/anthropic/index.js +166 -177
  9. package/dist/anthropic/index.js.map +1 -1
  10. package/dist/edge/index.d.ts +206 -194
  11. package/dist/edge/index.d.ts.map +1 -0
  12. package/dist/edge/index.js +1408 -1846
  13. package/dist/edge/index.js.map +1 -1
  14. package/dist/index.d.ts +265 -252
  15. package/dist/index.d.ts.map +1 -0
  16. package/dist/index.js +1349 -1761
  17. package/dist/index.js.map +1 -1
  18. package/dist/maas/edge/index.d.ts +62 -59
  19. package/dist/maas/edge/index.d.ts.map +1 -0
  20. package/dist/maas/edge/index.js +166 -196
  21. package/dist/maas/edge/index.js.map +1 -1
  22. package/dist/maas/index.d.ts +47 -45
  23. package/dist/maas/index.d.ts.map +1 -0
  24. package/dist/maas/index.js +101 -102
  25. package/dist/maas/index.js.map +1 -1
  26. package/dist/xai/edge/index.d.ts +77 -73
  27. package/dist/xai/edge/index.d.ts.map +1 -0
  28. package/dist/xai/edge/index.js +203 -235
  29. package/dist/xai/edge/index.js.map +1 -1
  30. package/dist/xai/index.d.ts +62 -59
  31. package/dist/xai/index.d.ts.map +1 -0
  32. package/dist/xai/index.js +138 -141
  33. package/dist/xai/index.js.map +1 -1
  34. package/docs/16-google-vertex.mdx +3 -3
  35. package/package.json +11 -11
  36. package/src/edge/google-vertex-auth-edge.ts +1 -1
  37. package/src/google-vertex-provider-base.ts +1 -1
package/dist/index.js CHANGED
@@ -1,1819 +1,1407 @@
1
- // src/google-vertex-provider.ts
2
- import { loadOptionalSetting as loadOptionalSetting2, resolve as resolve7 } from "@ai-sdk/provider-utils";
3
-
4
- // src/google-vertex-auth-google-auth-library.ts
1
+ import { WORKFLOW_DESERIALIZE, WORKFLOW_SERIALIZE, combineHeaders, connectToWebSocket, convertBase64ToUint8Array, convertToBase64, convertUint8ArrayToBase64, createJsonErrorResponseHandler, createJsonResponseHandler, generateId, lazySchema, loadOptionalSetting, loadSetting, normalizeHeaders, parseProviderOptions, postJsonToApi, resolve, safeParseJSON, serializeModelOptions, waitForWebSocketBufferDrain, withUserAgentSuffix, withoutTrailingSlash, zodSchema } from "@ai-sdk/provider-utils";
5
2
  import { GoogleAuth } from "google-auth-library";
3
+ import { GoogleInteractionsLanguageModel, GoogleLanguageModel, GoogleSpeechModel, googleTools } from "@ai-sdk/google/internal";
4
+ import { AISDKError, InvalidArgumentError, TooManyEmbeddingValuesForCallError } from "@ai-sdk/provider";
5
+ import { z } from "zod/v4";
6
+ //#region src/google-vertex-auth-google-auth-library.ts
6
7
  function createAuthTokenGenerator(options) {
7
- const auth = new GoogleAuth({
8
- scopes: ["https://www.googleapis.com/auth/cloud-platform"],
9
- ...options
10
- });
11
- return async function generateAuthToken() {
12
- const client = await auth.getClient();
13
- const token = await client.getAccessToken();
14
- return token?.token ?? null;
15
- };
8
+ const auth = new GoogleAuth({
9
+ scopes: ["https://www.googleapis.com/auth/cloud-platform"],
10
+ ...options
11
+ });
12
+ return async function generateAuthToken() {
13
+ return (await (await auth.getClient()).getAccessToken())?.token ?? null;
14
+ };
16
15
  }
17
-
18
- // src/google-vertex-provider-base.ts
19
- import {
20
- GoogleInteractionsLanguageModel,
21
- GoogleLanguageModel as GoogleLanguageModel2,
22
- GoogleSpeechModel
23
- } from "@ai-sdk/google/internal";
24
- import {
25
- generateId,
26
- loadOptionalSetting,
27
- loadSetting,
28
- normalizeHeaders,
29
- resolve as resolve6,
30
- withoutTrailingSlash,
31
- withUserAgentSuffix
32
- } from "@ai-sdk/provider-utils";
33
-
34
- // src/version.ts
35
- var VERSION = true ? "5.0.97" : "0.0.0-test";
36
-
37
- // src/google-vertex-embedding-model.ts
38
- import {
39
- TooManyEmbeddingValuesForCallError
40
- } from "@ai-sdk/provider";
41
- import {
42
- combineHeaders,
43
- createJsonResponseHandler,
44
- postJsonToApi,
45
- resolve,
46
- parseProviderOptions,
47
- serializeModelOptions,
48
- WORKFLOW_SERIALIZE,
49
- WORKFLOW_DESERIALIZE
50
- } from "@ai-sdk/provider-utils";
51
- import { z as z3 } from "zod/v4";
52
-
53
- // src/google-vertex-error.ts
54
- import { createJsonErrorResponseHandler } from "@ai-sdk/provider-utils";
55
- import { z } from "zod/v4";
56
- var googleVertexErrorDataSchema = z.object({
57
- error: z.object({
58
- code: z.number().nullable(),
59
- message: z.string(),
60
- status: z.string()
61
- })
16
+ //#endregion
17
+ //#region src/version.ts
18
+ const VERSION = "5.0.99";
19
+ //#endregion
20
+ //#region src/google-vertex-error.ts
21
+ const googleVertexErrorDataSchema = z.object({ error: z.object({
22
+ code: z.number().nullable(),
23
+ message: z.string(),
24
+ status: z.string()
25
+ }) });
26
+ const googleVertexFailedResponseHandler = createJsonErrorResponseHandler({
27
+ errorSchema: googleVertexErrorDataSchema,
28
+ errorToMessage: (data) => data.error.message
62
29
  });
63
- var googleVertexFailedResponseHandler = createJsonErrorResponseHandler(
64
- {
65
- errorSchema: googleVertexErrorDataSchema,
66
- errorToMessage: (data) => data.error.message
67
- }
68
- );
69
-
70
- // src/google-vertex-embedding-model-options.ts
71
- import { z as z2 } from "zod/v4";
72
- var googleVertexEmbeddingModelOptions = z2.object({
73
- /**
74
- * Optional. Optional reduced dimension for the output embedding.
75
- * If set, excessive values in the output embedding are truncated from the end.
76
- */
77
- outputDimensionality: z2.number().optional(),
78
- /**
79
- * Optional. Specifies the task type for generating embeddings.
80
- * Supported task types:
81
- * - SEMANTIC_SIMILARITY: Optimized for text similarity.
82
- * - CLASSIFICATION: Optimized for text classification.
83
- * - CLUSTERING: Optimized for clustering texts based on similarity.
84
- * - RETRIEVAL_DOCUMENT: Optimized for document retrieval.
85
- * - RETRIEVAL_QUERY: Optimized for query-based retrieval.
86
- * - QUESTION_ANSWERING: Optimized for answering questions.
87
- * - FACT_VERIFICATION: Optimized for verifying factual information.
88
- * - CODE_RETRIEVAL_QUERY: Optimized for retrieving code blocks based on natural language queries.
89
- */
90
- taskType: z2.enum([
91
- "SEMANTIC_SIMILARITY",
92
- "CLASSIFICATION",
93
- "CLUSTERING",
94
- "RETRIEVAL_DOCUMENT",
95
- "RETRIEVAL_QUERY",
96
- "QUESTION_ANSWERING",
97
- "FACT_VERIFICATION",
98
- "CODE_RETRIEVAL_QUERY"
99
- ]).optional(),
100
- /**
101
- * Optional. The title of the document being embedded.
102
- * Only valid when task_type is set to 'RETRIEVAL_DOCUMENT'.
103
- * Helps the model produce better embeddings by providing additional context.
104
- */
105
- title: z2.string().optional(),
106
- /**
107
- * Optional. When set to true, input text will be truncated. When set to false,
108
- * an error is returned if the input text is longer than the maximum length supported by the model. Defaults to true.
109
- */
110
- autoTruncate: z2.boolean().optional()
30
+ //#endregion
31
+ //#region src/google-vertex-embedding-model-options.ts
32
+ const googleVertexEmbeddingModelOptions = z.object({
33
+ /**
34
+ * Optional. Optional reduced dimension for the output embedding.
35
+ * If set, excessive values in the output embedding are truncated from the end.
36
+ */
37
+ outputDimensionality: z.number().optional(),
38
+ /**
39
+ * Optional. Specifies the task type for generating embeddings.
40
+ * Supported task types:
41
+ * - SEMANTIC_SIMILARITY: Optimized for text similarity.
42
+ * - CLASSIFICATION: Optimized for text classification.
43
+ * - CLUSTERING: Optimized for clustering texts based on similarity.
44
+ * - RETRIEVAL_DOCUMENT: Optimized for document retrieval.
45
+ * - RETRIEVAL_QUERY: Optimized for query-based retrieval.
46
+ * - QUESTION_ANSWERING: Optimized for answering questions.
47
+ * - FACT_VERIFICATION: Optimized for verifying factual information.
48
+ * - CODE_RETRIEVAL_QUERY: Optimized for retrieving code blocks based on natural language queries.
49
+ */
50
+ taskType: z.enum([
51
+ "SEMANTIC_SIMILARITY",
52
+ "CLASSIFICATION",
53
+ "CLUSTERING",
54
+ "RETRIEVAL_DOCUMENT",
55
+ "RETRIEVAL_QUERY",
56
+ "QUESTION_ANSWERING",
57
+ "FACT_VERIFICATION",
58
+ "CODE_RETRIEVAL_QUERY"
59
+ ]).optional(),
60
+ /**
61
+ * Optional. The title of the document being embedded.
62
+ * Only valid when task_type is set to 'RETRIEVAL_DOCUMENT'.
63
+ * Helps the model produce better embeddings by providing additional context.
64
+ */
65
+ title: z.string().optional(),
66
+ /**
67
+ * Optional. When set to true, input text will be truncated. When set to false,
68
+ * an error is returned if the input text is longer than the maximum length supported by the model. Defaults to true.
69
+ */
70
+ autoTruncate: z.boolean().optional()
111
71
  });
112
-
113
- // src/google-vertex-embedding-model.ts
114
- var GoogleVertexEmbeddingModel = class _GoogleVertexEmbeddingModel {
115
- constructor(modelId, config) {
116
- this.specificationVersion = "v4";
117
- this.supportsParallelCalls = true;
118
- this.modelId = modelId;
119
- this.config = config;
120
- }
121
- static [WORKFLOW_SERIALIZE](model) {
122
- return serializeModelOptions({
123
- modelId: model.modelId,
124
- config: model.config
125
- });
126
- }
127
- static [WORKFLOW_DESERIALIZE](options) {
128
- return new _GoogleVertexEmbeddingModel(options.modelId, options.config);
129
- }
130
- get provider() {
131
- return this.config.provider;
132
- }
133
- // gemini-embedding-2 models only support :embedContent (one value per call),
134
- // not the :predict batch endpoint. https://github.com/vercel/ai/issues/15853
135
- get maxEmbeddingsPerCall() {
136
- return usesEmbedContentEndpoint(this.modelId) ? 1 : 250;
137
- }
138
- async doEmbed({
139
- values,
140
- headers,
141
- abortSignal,
142
- providerOptions
143
- }) {
144
- let googleOptions = await parseProviderOptions({
145
- provider: "googleVertex",
146
- providerOptions,
147
- schema: googleVertexEmbeddingModelOptions
148
- });
149
- if (googleOptions == null) {
150
- googleOptions = await parseProviderOptions({
151
- provider: "vertex",
152
- providerOptions,
153
- schema: googleVertexEmbeddingModelOptions
154
- });
155
- }
156
- if (googleOptions == null) {
157
- googleOptions = await parseProviderOptions({
158
- provider: "google",
159
- providerOptions,
160
- schema: googleVertexEmbeddingModelOptions
161
- });
162
- }
163
- googleOptions = googleOptions ?? {};
164
- if (values.length > this.maxEmbeddingsPerCall) {
165
- throw new TooManyEmbeddingValuesForCallError({
166
- provider: this.provider,
167
- modelId: this.modelId,
168
- maxEmbeddingsPerCall: this.maxEmbeddingsPerCall,
169
- values
170
- });
171
- }
172
- const mergedHeaders = combineHeaders(
173
- this.config.headers ? await resolve(this.config.headers) : void 0,
174
- headers
175
- );
176
- if (usesEmbedContentEndpoint(this.modelId)) {
177
- const {
178
- responseHeaders: responseHeaders2,
179
- value: response2,
180
- rawValue: rawValue2
181
- } = await postJsonToApi({
182
- url: `${this.config.baseURL}/models/${this.modelId}:embedContent`,
183
- headers: mergedHeaders,
184
- body: {
185
- content: { parts: [{ text: values[0] }] },
186
- embedContentConfig: {
187
- outputDimensionality: googleOptions.outputDimensionality,
188
- taskType: googleOptions.taskType,
189
- title: googleOptions.title,
190
- autoTruncate: googleOptions.autoTruncate
191
- }
192
- },
193
- failedResponseHandler: googleVertexFailedResponseHandler,
194
- successfulResponseHandler: createJsonResponseHandler(
195
- googleVertexEmbedContentResponseSchema
196
- ),
197
- abortSignal,
198
- fetch: this.config.fetch
199
- });
200
- return {
201
- warnings: [],
202
- embeddings: [response2.embedding.values],
203
- usage: response2.usageMetadata?.promptTokenCount == null ? void 0 : { tokens: response2.usageMetadata.promptTokenCount },
204
- response: { headers: responseHeaders2, body: rawValue2 }
205
- };
206
- }
207
- const url = `${this.config.baseURL}/models/${this.modelId}:predict`;
208
- const {
209
- responseHeaders,
210
- value: response,
211
- rawValue
212
- } = await postJsonToApi({
213
- url,
214
- headers: mergedHeaders,
215
- body: {
216
- instances: values.map((value) => ({
217
- content: value,
218
- task_type: googleOptions.taskType,
219
- title: googleOptions.title
220
- })),
221
- parameters: {
222
- outputDimensionality: googleOptions.outputDimensionality,
223
- autoTruncate: googleOptions.autoTruncate
224
- }
225
- },
226
- failedResponseHandler: googleVertexFailedResponseHandler,
227
- successfulResponseHandler: createJsonResponseHandler(
228
- googleVertexTextEmbeddingResponseSchema
229
- ),
230
- abortSignal,
231
- fetch: this.config.fetch
232
- });
233
- return {
234
- warnings: [],
235
- embeddings: response.predictions.map(
236
- (prediction) => prediction.embeddings.values
237
- ),
238
- usage: {
239
- tokens: response.predictions.reduce(
240
- (tokenCount, prediction) => tokenCount + prediction.embeddings.statistics.token_count,
241
- 0
242
- )
243
- },
244
- response: { headers: responseHeaders, body: rawValue }
245
- };
246
- }
72
+ //#endregion
73
+ //#region src/google-vertex-embedding-model.ts
74
+ var GoogleVertexEmbeddingModel = class GoogleVertexEmbeddingModel {
75
+ static [WORKFLOW_SERIALIZE](model) {
76
+ return serializeModelOptions({
77
+ modelId: model.modelId,
78
+ config: model.config
79
+ });
80
+ }
81
+ static [WORKFLOW_DESERIALIZE](options) {
82
+ return new GoogleVertexEmbeddingModel(options.modelId, options.config);
83
+ }
84
+ get provider() {
85
+ return this.config.provider;
86
+ }
87
+ get maxEmbeddingsPerCall() {
88
+ return usesEmbedContentEndpoint(this.modelId) ? 1 : 250;
89
+ }
90
+ constructor(modelId, config) {
91
+ this.specificationVersion = "v4";
92
+ this.supportsParallelCalls = true;
93
+ this.modelId = modelId;
94
+ this.config = config;
95
+ }
96
+ async doEmbed({ values, headers, abortSignal, providerOptions }) {
97
+ let googleOptions = await parseProviderOptions({
98
+ provider: "googleVertex",
99
+ providerOptions,
100
+ schema: googleVertexEmbeddingModelOptions
101
+ });
102
+ if (googleOptions == null) googleOptions = await parseProviderOptions({
103
+ provider: "vertex",
104
+ providerOptions,
105
+ schema: googleVertexEmbeddingModelOptions
106
+ });
107
+ if (googleOptions == null) googleOptions = await parseProviderOptions({
108
+ provider: "google",
109
+ providerOptions,
110
+ schema: googleVertexEmbeddingModelOptions
111
+ });
112
+ googleOptions = googleOptions ?? {};
113
+ if (values.length > this.maxEmbeddingsPerCall) throw new TooManyEmbeddingValuesForCallError({
114
+ provider: this.provider,
115
+ modelId: this.modelId,
116
+ maxEmbeddingsPerCall: this.maxEmbeddingsPerCall,
117
+ values
118
+ });
119
+ const mergedHeaders = combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, headers);
120
+ if (usesEmbedContentEndpoint(this.modelId)) {
121
+ const { responseHeaders, value: response, rawValue } = await postJsonToApi({
122
+ url: `${this.config.baseURL}/models/${this.modelId}:embedContent`,
123
+ headers: mergedHeaders,
124
+ body: {
125
+ content: { parts: [{ text: values[0] }] },
126
+ embedContentConfig: {
127
+ outputDimensionality: googleOptions.outputDimensionality,
128
+ taskType: googleOptions.taskType,
129
+ title: googleOptions.title,
130
+ autoTruncate: googleOptions.autoTruncate
131
+ }
132
+ },
133
+ failedResponseHandler: googleVertexFailedResponseHandler,
134
+ successfulResponseHandler: createJsonResponseHandler(googleVertexEmbedContentResponseSchema),
135
+ abortSignal,
136
+ fetch: this.config.fetch
137
+ });
138
+ return {
139
+ warnings: [],
140
+ embeddings: [response.embedding.values],
141
+ usage: response.usageMetadata?.promptTokenCount == null ? void 0 : { tokens: response.usageMetadata.promptTokenCount },
142
+ response: {
143
+ headers: responseHeaders,
144
+ body: rawValue
145
+ }
146
+ };
147
+ }
148
+ const url = `${this.config.baseURL}/models/${this.modelId}:predict`;
149
+ const { responseHeaders, value: response, rawValue } = await postJsonToApi({
150
+ url,
151
+ headers: mergedHeaders,
152
+ body: {
153
+ instances: values.map((value) => ({
154
+ content: value,
155
+ task_type: googleOptions.taskType,
156
+ title: googleOptions.title
157
+ })),
158
+ parameters: {
159
+ outputDimensionality: googleOptions.outputDimensionality,
160
+ autoTruncate: googleOptions.autoTruncate
161
+ }
162
+ },
163
+ failedResponseHandler: googleVertexFailedResponseHandler,
164
+ successfulResponseHandler: createJsonResponseHandler(googleVertexTextEmbeddingResponseSchema),
165
+ abortSignal,
166
+ fetch: this.config.fetch
167
+ });
168
+ return {
169
+ warnings: [],
170
+ embeddings: response.predictions.map((prediction) => prediction.embeddings.values),
171
+ usage: { tokens: response.predictions.reduce((tokenCount, prediction) => tokenCount + prediction.embeddings.statistics.token_count, 0) },
172
+ response: {
173
+ headers: responseHeaders,
174
+ body: rawValue
175
+ }
176
+ };
177
+ }
247
178
  };
248
- var googleVertexTextEmbeddingResponseSchema = z3.object({
249
- predictions: z3.array(
250
- z3.object({
251
- embeddings: z3.object({
252
- values: z3.array(z3.number()),
253
- statistics: z3.object({
254
- token_count: z3.number()
255
- })
256
- })
257
- })
258
- )
259
- });
260
- var googleVertexEmbedContentResponseSchema = z3.object({
261
- embedding: z3.object({
262
- values: z3.array(z3.number())
263
- }),
264
- usageMetadata: z3.object({
265
- promptTokenCount: z3.number().nullish()
266
- }).nullish()
179
+ const googleVertexTextEmbeddingResponseSchema = z.object({ predictions: z.array(z.object({ embeddings: z.object({
180
+ values: z.array(z.number()),
181
+ statistics: z.object({ token_count: z.number() })
182
+ }) })) });
183
+ const googleVertexEmbedContentResponseSchema = z.object({
184
+ embedding: z.object({ values: z.array(z.number()) }),
185
+ usageMetadata: z.object({ promptTokenCount: z.number().nullish() }).nullish()
267
186
  });
268
187
  function usesEmbedContentEndpoint(modelId) {
269
- return modelId === "gemini-embedding-2" || modelId === "gemini-embedding-2-preview";
188
+ return modelId === "gemini-embedding-2" || modelId === "gemini-embedding-2-preview";
270
189
  }
271
-
272
- // src/google-vertex-image-model.ts
273
- import { GoogleLanguageModel } from "@ai-sdk/google/internal";
274
- import {
275
- convertToBase64,
276
- generateId as defaultGenerateId,
277
- serializeModelOptions as serializeModelOptions2,
278
- WORKFLOW_SERIALIZE as WORKFLOW_SERIALIZE2,
279
- WORKFLOW_DESERIALIZE as WORKFLOW_DESERIALIZE2
280
- } from "@ai-sdk/provider-utils";
281
- var googleVertexImageModelsWithFileInputSupport = /* @__PURE__ */ new Set([
282
- "gemini-2.5-flash-image",
283
- "gemini-3-pro-image-preview",
284
- "gemini-3.1-flash-image-preview"
190
+ //#endregion
191
+ //#region src/google-vertex-image-model.ts
192
+ const googleVertexImageModelsWithFileInputSupport = /* @__PURE__ */ new Set([
193
+ "gemini-2.5-flash-image",
194
+ "gemini-3-pro-image-preview",
195
+ "gemini-3.1-flash-image-preview"
285
196
  ]);
286
- var GoogleVertexImageModel = class _GoogleVertexImageModel {
287
- constructor(modelId, config) {
288
- this.modelId = modelId;
289
- this.config = config;
290
- this.specificationVersion = "v4";
291
- this.maxImagesPerCall = 1;
292
- }
293
- static [WORKFLOW_SERIALIZE2](model) {
294
- return serializeModelOptions2({
295
- modelId: model.modelId,
296
- config: model.config
297
- });
298
- }
299
- static [WORKFLOW_DESERIALIZE2](options) {
300
- return new _GoogleVertexImageModel(options.modelId, options.config);
301
- }
302
- get supportsFileInputs() {
303
- return googleVertexImageModelsWithFileInputSupport.has(this.modelId) ? true : void 0;
304
- }
305
- get supportsMaskInputs() {
306
- return this.supportsFileInputs === true ? false : void 0;
307
- }
308
- get provider() {
309
- return this.config.provider;
310
- }
311
- async doGenerate(options) {
312
- if (!this.modelId.startsWith("gemini-")) {
313
- throw new Error(
314
- "Google image models other than Gemini are no longer supported. Use a model ID that starts with `gemini-`."
315
- );
316
- }
317
- const {
318
- prompt,
319
- size,
320
- aspectRatio,
321
- seed,
322
- providerOptions,
323
- headers,
324
- abortSignal,
325
- files,
326
- mask
327
- } = options;
328
- const warnings = [];
329
- if (mask != null) {
330
- throw new Error(
331
- "Gemini image models do not support mask-based image editing."
332
- );
333
- }
334
- if (size != null) {
335
- warnings.push({
336
- type: "unsupported",
337
- feature: "size",
338
- details: "This model does not support the `size` option. Use `aspectRatio` instead."
339
- });
340
- }
341
- const userContent = [];
342
- if (prompt != null) {
343
- userContent.push({ type: "text", text: prompt });
344
- }
345
- if (files != null && files.length > 0) {
346
- for (const file of files) {
347
- if (file.type === "url") {
348
- userContent.push({
349
- type: "file",
350
- data: { type: "url", url: new URL(file.url) },
351
- mediaType: "image/*"
352
- });
353
- } else {
354
- userContent.push({
355
- type: "file",
356
- data: {
357
- type: "data",
358
- data: typeof file.data === "string" ? file.data : new Uint8Array(file.data)
359
- },
360
- mediaType: file.mediaType
361
- });
362
- }
363
- }
364
- }
365
- const languageModelPrompt = [
366
- { role: "user", content: userContent }
367
- ];
368
- const languageModel = new GoogleLanguageModel(this.modelId, {
369
- provider: this.config.provider,
370
- baseURL: this.config.baseURL,
371
- headers: this.config.headers ?? {},
372
- fetch: this.config.fetch,
373
- generateId: this.config.generateId ?? defaultGenerateId,
374
- supportedUrls: () => ({
375
- "*": [/^https?:\/\/.*$/, /^gs:\/\/.*$/]
376
- })
377
- });
378
- const {
379
- responseModalities: _strippedResponseModalities,
380
- imageConfig: userImageConfig,
381
- ...userVertexOptions
382
- } = providerOptions?.googleVertex ?? providerOptions?.vertex ?? {};
383
- const innerVertexOptions = {
384
- ...userVertexOptions,
385
- responseModalities: ["IMAGE"],
386
- imageConfig: aspectRatio != null || userImageConfig != null ? {
387
- ...userImageConfig,
388
- ...aspectRatio != null ? {
389
- aspectRatio
390
- } : {}
391
- } : void 0
392
- };
393
- const result = await languageModel.doGenerate({
394
- prompt: languageModelPrompt,
395
- seed,
396
- providerOptions: {
397
- googleVertex: innerVertexOptions,
398
- vertex: innerVertexOptions
399
- },
400
- headers,
401
- abortSignal
402
- });
403
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
404
- const images = [];
405
- for (const part of result.content) {
406
- if (part.type === "file" && part.mediaType.startsWith("image/") && part.data.type === "data") {
407
- images.push(convertToBase64(part.data.data));
408
- }
409
- }
410
- const geminiPayload = {
411
- images: images.map(() => ({}))
412
- };
413
- return {
414
- images,
415
- ...result.finishReason.unified === "content-filter" ? { isRetryable: false } : {},
416
- warnings,
417
- providerMetadata: {
418
- googleVertex: geminiPayload,
419
- vertex: geminiPayload
420
- },
421
- response: {
422
- timestamp: currentDate,
423
- modelId: this.modelId,
424
- headers: result.response?.headers
425
- },
426
- usage: result.usage ? {
427
- inputTokens: result.usage.inputTokens.total,
428
- outputTokens: result.usage.outputTokens.total,
429
- totalTokens: (result.usage.inputTokens.total ?? 0) + (result.usage.outputTokens.total ?? 0)
430
- } : void 0
431
- };
432
- }
197
+ var GoogleVertexImageModel = class GoogleVertexImageModel {
198
+ static [WORKFLOW_SERIALIZE](model) {
199
+ return serializeModelOptions({
200
+ modelId: model.modelId,
201
+ config: model.config
202
+ });
203
+ }
204
+ static [WORKFLOW_DESERIALIZE](options) {
205
+ return new GoogleVertexImageModel(options.modelId, options.config);
206
+ }
207
+ get supportsFileInputs() {
208
+ return googleVertexImageModelsWithFileInputSupport.has(this.modelId) ? true : void 0;
209
+ }
210
+ get supportsMaskInputs() {
211
+ return this.supportsFileInputs === true ? false : void 0;
212
+ }
213
+ get provider() {
214
+ return this.config.provider;
215
+ }
216
+ constructor(modelId, config) {
217
+ this.modelId = modelId;
218
+ this.config = config;
219
+ this.specificationVersion = "v4";
220
+ this.maxImagesPerCall = 1;
221
+ }
222
+ async doGenerate(options) {
223
+ if (!this.modelId.startsWith("gemini-")) throw new Error("Google image models other than Gemini are no longer supported. Use a model ID that starts with `gemini-`.");
224
+ const { prompt, size, aspectRatio, seed, providerOptions, headers, abortSignal, files, mask } = options;
225
+ const warnings = [];
226
+ if (mask != null) throw new Error("Gemini image models do not support mask-based image editing.");
227
+ if (size != null) warnings.push({
228
+ type: "unsupported",
229
+ feature: "size",
230
+ details: "This model does not support the `size` option. Use `aspectRatio` instead."
231
+ });
232
+ const userContent = [];
233
+ if (prompt != null) userContent.push({
234
+ type: "text",
235
+ text: prompt
236
+ });
237
+ if (files != null && files.length > 0) for (const file of files) if (file.type === "url") userContent.push({
238
+ type: "file",
239
+ data: {
240
+ type: "url",
241
+ url: new URL(file.url)
242
+ },
243
+ mediaType: "image/*"
244
+ });
245
+ else userContent.push({
246
+ type: "file",
247
+ data: {
248
+ type: "data",
249
+ data: typeof file.data === "string" ? file.data : new Uint8Array(file.data)
250
+ },
251
+ mediaType: file.mediaType
252
+ });
253
+ const languageModelPrompt = [{
254
+ role: "user",
255
+ content: userContent
256
+ }];
257
+ const languageModel = new GoogleLanguageModel(this.modelId, {
258
+ provider: this.config.provider,
259
+ baseURL: this.config.baseURL,
260
+ headers: this.config.headers ?? {},
261
+ fetch: this.config.fetch,
262
+ generateId: this.config.generateId ?? generateId,
263
+ supportedUrls: () => ({ "*": [/^https?:\/\/.*$/, /^gs:\/\/.*$/] })
264
+ });
265
+ const { responseModalities: _strippedResponseModalities, imageConfig: userImageConfig, ...userVertexOptions } = providerOptions?.googleVertex ?? providerOptions?.vertex ?? {};
266
+ const innerVertexOptions = {
267
+ ...userVertexOptions,
268
+ responseModalities: ["IMAGE"],
269
+ imageConfig: aspectRatio != null || userImageConfig != null ? {
270
+ ...userImageConfig,
271
+ ...aspectRatio != null ? { aspectRatio } : {}
272
+ } : void 0
273
+ };
274
+ const result = await languageModel.doGenerate({
275
+ prompt: languageModelPrompt,
276
+ seed,
277
+ providerOptions: {
278
+ googleVertex: innerVertexOptions,
279
+ vertex: innerVertexOptions
280
+ },
281
+ headers,
282
+ abortSignal
283
+ });
284
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
285
+ const images = [];
286
+ for (const part of result.content) if (part.type === "file" && part.mediaType.startsWith("image/") && part.data.type === "data") images.push(convertToBase64(part.data.data));
287
+ const geminiPayload = { images: images.map(() => ({})) };
288
+ return {
289
+ images,
290
+ ...result.finishReason.unified === "content-filter" ? { isRetryable: false } : {},
291
+ warnings,
292
+ providerMetadata: {
293
+ googleVertex: geminiPayload,
294
+ vertex: geminiPayload
295
+ },
296
+ response: {
297
+ timestamp: currentDate,
298
+ modelId: this.modelId,
299
+ headers: result.response?.headers
300
+ },
301
+ usage: result.usage ? {
302
+ inputTokens: result.usage.inputTokens.total,
303
+ outputTokens: result.usage.outputTokens.total,
304
+ totalTokens: (result.usage.inputTokens.total ?? 0) + (result.usage.outputTokens.total ?? 0)
305
+ } : void 0
306
+ };
307
+ }
433
308
  };
434
-
435
- // src/google-vertex-cloud-tts-speech-model.ts
436
- import {
437
- combineHeaders as combineHeaders2,
438
- convertBase64ToUint8Array,
439
- createJsonResponseHandler as createJsonResponseHandler2,
440
- postJsonToApi as postJsonToApi2,
441
- resolve as resolve2,
442
- serializeModelOptions as serializeModelOptions3,
443
- WORKFLOW_DESERIALIZE as WORKFLOW_DESERIALIZE3,
444
- WORKFLOW_SERIALIZE as WORKFLOW_SERIALIZE3
445
- } from "@ai-sdk/provider-utils";
446
- import { z as z4 } from "zod/v4";
447
- var DEFAULT_VOICE = "Kore";
448
- var DEFAULT_LANGUAGE = "en-US";
449
- var CHIRP3_HD_VOICE_INFIX = "Chirp3-HD";
450
- var CLOUD_TTS_SYNTHESIZE_URL = "https://texttospeech.googleapis.com/v1/text:synthesize";
451
- var GoogleVertexCloudTTSSpeechModel = class _GoogleVertexCloudTTSSpeechModel {
452
- constructor(modelId, config) {
453
- this.modelId = modelId;
454
- this.config = config;
455
- this.specificationVersion = "v4";
456
- }
457
- static [WORKFLOW_SERIALIZE3](model) {
458
- return serializeModelOptions3({
459
- modelId: model.modelId,
460
- config: model.config
461
- });
462
- }
463
- static [WORKFLOW_DESERIALIZE3](options) {
464
- return new _GoogleVertexCloudTTSSpeechModel(options.modelId, options.config);
465
- }
466
- get provider() {
467
- return this.config.provider;
468
- }
469
- async doGenerate(options) {
470
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
471
- const warnings = [];
472
- const {
473
- text,
474
- voice = DEFAULT_VOICE,
475
- outputFormat,
476
- instructions,
477
- speed,
478
- language
479
- } = options;
480
- let voiceName;
481
- let languageCode;
482
- if (voice.includes(CHIRP3_HD_VOICE_INFIX)) {
483
- voiceName = voice;
484
- const localePrefix = voice.split(CHIRP3_HD_VOICE_INFIX)[0].replace(/-$/, "");
485
- languageCode = language ?? (localePrefix || DEFAULT_LANGUAGE);
486
- } else {
487
- languageCode = language ?? DEFAULT_LANGUAGE;
488
- voiceName = `${languageCode}-${CHIRP3_HD_VOICE_INFIX}-${voice}`;
489
- }
490
- if (instructions != null) {
491
- warnings.push({
492
- type: "unsupported",
493
- feature: "instructions",
494
- details: "Google Cloud Text-to-Speech Chirp 3: HD voices do not support the `instructions` option. It was ignored."
495
- });
496
- }
497
- if (outputFormat != null && outputFormat !== "wav") {
498
- warnings.push({
499
- type: "unsupported",
500
- feature: "outputFormat",
501
- details: `Unsupported output format: ${outputFormat}. Using wav instead.`
502
- });
503
- }
504
- const requestBody = {
505
- input: { text },
506
- voice: { languageCode, name: voiceName },
507
- audioConfig: {
508
- audioEncoding: "LINEAR16",
509
- ...speed != null ? { speakingRate: speed } : {}
510
- }
511
- };
512
- const {
513
- value: response,
514
- responseHeaders,
515
- rawValue: rawResponse
516
- } = await postJsonToApi2({
517
- url: CLOUD_TTS_SYNTHESIZE_URL,
518
- headers: combineHeaders2(
519
- this.config.headers ? await resolve2(this.config.headers) : void 0,
520
- options.headers
521
- ),
522
- body: requestBody,
523
- failedResponseHandler: googleVertexFailedResponseHandler,
524
- successfulResponseHandler: createJsonResponseHandler2(
525
- googleVertexCloudTTSResponseSchema
526
- ),
527
- abortSignal: options.abortSignal,
528
- fetch: this.config.fetch
529
- });
530
- const audio = response.audioContent != null ? convertBase64ToUint8Array(response.audioContent) : new Uint8Array(0);
531
- return {
532
- audio,
533
- warnings,
534
- request: {
535
- body: JSON.stringify(requestBody)
536
- },
537
- response: {
538
- timestamp: currentDate,
539
- modelId: this.modelId,
540
- headers: responseHeaders,
541
- body: rawResponse
542
- },
543
- providerMetadata: {
544
- google: {
545
- mimeType: "audio/wav"
546
- }
547
- }
548
- };
549
- }
309
+ //#endregion
310
+ //#region src/google-vertex-cloud-tts-speech-model.ts
311
+ const DEFAULT_VOICE = "Kore";
312
+ const DEFAULT_LANGUAGE = "en-US";
313
+ const CHIRP3_HD_VOICE_INFIX = "Chirp3-HD";
314
+ const CLOUD_TTS_SYNTHESIZE_URL = "https://texttospeech.googleapis.com/v1/text:synthesize";
315
+ /**
316
+ * Speech model for Chirp 3: HD voices on the Google Cloud Text-to-Speech API.
317
+ *
318
+ * Unlike the Gemini TTS models (which go through the Vertex
319
+ * `generateContent` endpoint via `GoogleSpeechModel`), Chirp 3: HD voices are
320
+ * served by the dedicated Cloud Text-to-Speech `text:synthesize` endpoint,
321
+ * reusing the provider's Google Cloud credentials.
322
+ */
323
+ var GoogleVertexCloudTTSSpeechModel = class GoogleVertexCloudTTSSpeechModel {
324
+ static [WORKFLOW_SERIALIZE](model) {
325
+ return serializeModelOptions({
326
+ modelId: model.modelId,
327
+ config: model.config
328
+ });
329
+ }
330
+ static [WORKFLOW_DESERIALIZE](options) {
331
+ return new GoogleVertexCloudTTSSpeechModel(options.modelId, options.config);
332
+ }
333
+ get provider() {
334
+ return this.config.provider;
335
+ }
336
+ constructor(modelId, config) {
337
+ this.modelId = modelId;
338
+ this.config = config;
339
+ this.specificationVersion = "v4";
340
+ }
341
+ async doGenerate(options) {
342
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
343
+ const warnings = [];
344
+ const { text, voice = DEFAULT_VOICE, outputFormat, instructions, speed, language } = options;
345
+ let voiceName;
346
+ let languageCode;
347
+ if (voice.includes(CHIRP3_HD_VOICE_INFIX)) {
348
+ voiceName = voice;
349
+ const localePrefix = voice.split(CHIRP3_HD_VOICE_INFIX)[0].replace(/-$/, "");
350
+ languageCode = language ?? (localePrefix || DEFAULT_LANGUAGE);
351
+ } else {
352
+ languageCode = language ?? DEFAULT_LANGUAGE;
353
+ voiceName = `${languageCode}-${CHIRP3_HD_VOICE_INFIX}-${voice}`;
354
+ }
355
+ if (instructions != null) warnings.push({
356
+ type: "unsupported",
357
+ feature: "instructions",
358
+ details: "Google Cloud Text-to-Speech Chirp 3: HD voices do not support the `instructions` option. It was ignored."
359
+ });
360
+ if (outputFormat != null && outputFormat !== "wav") warnings.push({
361
+ type: "unsupported",
362
+ feature: "outputFormat",
363
+ details: `Unsupported output format: ${outputFormat}. Using wav instead.`
364
+ });
365
+ const requestBody = {
366
+ input: { text },
367
+ voice: {
368
+ languageCode,
369
+ name: voiceName
370
+ },
371
+ audioConfig: {
372
+ audioEncoding: "LINEAR16",
373
+ ...speed != null ? { speakingRate: speed } : {}
374
+ }
375
+ };
376
+ const { value: response, responseHeaders, rawValue: rawResponse } = await postJsonToApi({
377
+ url: CLOUD_TTS_SYNTHESIZE_URL,
378
+ headers: combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, options.headers),
379
+ body: requestBody,
380
+ failedResponseHandler: googleVertexFailedResponseHandler,
381
+ successfulResponseHandler: createJsonResponseHandler(googleVertexCloudTTSResponseSchema),
382
+ abortSignal: options.abortSignal,
383
+ fetch: this.config.fetch
384
+ });
385
+ return {
386
+ audio: response.audioContent != null ? convertBase64ToUint8Array(response.audioContent) : /* @__PURE__ */ new Uint8Array(0),
387
+ warnings,
388
+ request: { body: JSON.stringify(requestBody) },
389
+ response: {
390
+ timestamp: currentDate,
391
+ modelId: this.modelId,
392
+ headers: responseHeaders,
393
+ body: rawResponse
394
+ },
395
+ providerMetadata: { google: { mimeType: "audio/wav" } }
396
+ };
397
+ }
550
398
  };
551
- var googleVertexCloudTTSResponseSchema = z4.object({
552
- audioContent: z4.string().nullish()
553
- });
554
-
555
- // src/google-vertex-tools.ts
556
- import { googleTools } from "@ai-sdk/google/internal";
557
- var googleVertexTools = {
558
- googleSearch: googleTools.googleSearch,
559
- enterpriseWebSearch: googleTools.enterpriseWebSearch,
560
- googleMaps: googleTools.googleMaps,
561
- urlContext: googleTools.urlContext,
562
- fileSearch: googleTools.fileSearch,
563
- codeExecution: googleTools.codeExecution,
564
- vertexRagStore: googleTools.vertexRagStore
399
+ const googleVertexCloudTTSResponseSchema = z.object({ audioContent: z.string().nullish() });
400
+ //#endregion
401
+ //#region src/google-vertex-tools.ts
402
+ const googleVertexTools = {
403
+ googleSearch: googleTools.googleSearch,
404
+ enterpriseWebSearch: googleTools.enterpriseWebSearch,
405
+ googleMaps: googleTools.googleMaps,
406
+ urlContext: googleTools.urlContext,
407
+ fileSearch: googleTools.fileSearch,
408
+ codeExecution: googleTools.codeExecution,
409
+ vertexRagStore: googleTools.vertexRagStore
565
410
  };
566
-
567
- // src/google-vertex-transcription-model.ts
568
- import {
569
- combineHeaders as combineHeaders3,
570
- convertUint8ArrayToBase64,
571
- createJsonResponseHandler as createJsonResponseHandler3,
572
- parseProviderOptions as parseProviderOptions2,
573
- postJsonToApi as postJsonToApi3,
574
- resolve as resolve3,
575
- serializeModelOptions as serializeModelOptions4,
576
- WORKFLOW_DESERIALIZE as WORKFLOW_DESERIALIZE4,
577
- WORKFLOW_SERIALIZE as WORKFLOW_SERIALIZE4
578
- } from "@ai-sdk/provider-utils";
579
- import { z as z6 } from "zod/v4";
580
-
581
- // src/google-vertex-transcription-model-options.ts
582
- import { z as z5 } from "zod/v4";
583
- var googleVertexTranscriptionProviderOptionsSchema = z5.object({
584
- /**
585
- * BCP-47 language codes to recognize (e.g. `['en-US']`), or `['auto']` to let
586
- * Chirp auto-detect the spoken language. Defaults to `['auto']`. For
587
- * `telephony`, pass a supported explicit language code.
588
- */
589
- languageCodes: z5.array(z5.string()).optional(),
590
- /**
591
- * Whether to add punctuation to the transcript. Defaults to `true`.
592
- */
593
- enableAutomaticPunctuation: z5.boolean().optional(),
594
- /**
595
- * Whether to include word-level timestamps. Defaults to `true` so the
596
- * transcription result can include segments.
597
- *
598
- * Enabling word-level timestamps can reduce transcription quality and speed
599
- * for Chirp models.
600
- */
601
- enableWordTimeOffsets: z5.boolean().optional(),
602
- /**
603
- * The Cloud Speech-to-Text region for the request (e.g. `'us'`, `'eu'`,
604
- * `'us-central1'`). Defaults to the provider `location`.
605
- *
606
- * Note: Speech-to-Text regions differ from Vertex AI regions. Chirp is only
607
- * available in specific Speech-to-Text regions and is not available in the
608
- * `global` location.
609
- */
610
- region: z5.string().optional()
411
+ //#endregion
412
+ //#region src/google-vertex-transcription-model-options.ts
413
+ const googleVertexTranscriptionProviderOptionsSchema = z.object({
414
+ /**
415
+ * BCP-47 language codes to recognize (e.g. `['en-US']`), or `['auto']` to let
416
+ * Chirp auto-detect the spoken language. Defaults to `['auto']`. For
417
+ * `telephony`, pass a supported explicit language code.
418
+ */
419
+ languageCodes: z.array(z.string()).optional(),
420
+ /**
421
+ * Whether to add punctuation to the transcript. Defaults to `true`.
422
+ */
423
+ enableAutomaticPunctuation: z.boolean().optional(),
424
+ /**
425
+ * Whether to include word-level timestamps. Defaults to `true` so the
426
+ * transcription result can include segments.
427
+ *
428
+ * Enabling word-level timestamps can reduce transcription quality and speed
429
+ * for Chirp models.
430
+ */
431
+ enableWordTimeOffsets: z.boolean().optional(),
432
+ /**
433
+ * The Cloud Speech-to-Text region for the request (e.g. `'us'`, `'eu'`,
434
+ * `'us-central1'`). Defaults to the provider `location`.
435
+ *
436
+ * Note: Speech-to-Text regions differ from Vertex AI regions. Chirp is only
437
+ * available in specific Speech-to-Text regions and is not available in the
438
+ * `global` location.
439
+ */
440
+ region: z.string().optional()
611
441
  });
612
-
613
- // src/google-vertex-transcription-model.ts
442
+ //#endregion
443
+ //#region src/google-vertex-transcription-model.ts
614
444
  function parseDurationSeconds(value) {
615
- if (value == null) {
616
- return void 0;
617
- }
618
- const seconds = Number.parseFloat(value);
619
- return Number.isFinite(seconds) ? seconds : void 0;
445
+ if (value == null) return;
446
+ const seconds = Number.parseFloat(value);
447
+ return Number.isFinite(seconds) ? seconds : void 0;
620
448
  }
621
449
  function convertBcp47ToIso6391(value) {
622
- if (value == null) {
623
- return void 0;
624
- }
625
- try {
626
- const language = new Intl.Locale(value).language;
627
- return language.length === 2 ? language : void 0;
628
- } catch {
629
- return void 0;
630
- }
450
+ if (value == null) return;
451
+ try {
452
+ const language = new Intl.Locale(value).language;
453
+ return language.length === 2 ? language : void 0;
454
+ } catch {
455
+ return;
456
+ }
631
457
  }
632
- var GoogleVertexTranscriptionModel = class _GoogleVertexTranscriptionModel {
633
- constructor(modelId, config) {
634
- this.modelId = modelId;
635
- this.config = config;
636
- this.specificationVersion = "v4";
637
- }
638
- static [WORKFLOW_SERIALIZE4](model) {
639
- return serializeModelOptions4({
640
- modelId: model.modelId,
641
- config: model.config
642
- });
643
- }
644
- static [WORKFLOW_DESERIALIZE4](options) {
645
- return new _GoogleVertexTranscriptionModel(options.modelId, options.config);
646
- }
647
- get provider() {
648
- return this.config.provider;
649
- }
650
- async doGenerate(options) {
651
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
652
- const warnings = [];
653
- let googleOptions;
654
- for (const provider of ["googleVertex", "vertex", "google"]) {
655
- googleOptions = await parseProviderOptions2({
656
- provider,
657
- providerOptions: options.providerOptions,
658
- schema: googleVertexTranscriptionProviderOptionsSchema
659
- });
660
- if (googleOptions != null) {
661
- break;
662
- }
663
- }
664
- const region = googleOptions?.region ?? this.config.location;
665
- const languageCodes = googleOptions?.languageCodes ?? ["auto"];
666
- const content = typeof options.audio === "string" ? options.audio : convertUint8ArrayToBase64(options.audio);
667
- const requestBody = {
668
- config: {
669
- model: this.modelId,
670
- languageCodes,
671
- // Let Speech-to-Text auto-detect the audio encoding (wav/mp3/flac/…).
672
- autoDecodingConfig: {},
673
- features: {
674
- // Word timing populates `segments`.
675
- enableWordTimeOffsets: googleOptions?.enableWordTimeOffsets ?? true,
676
- enableAutomaticPunctuation: googleOptions?.enableAutomaticPunctuation ?? true
677
- }
678
- },
679
- content
680
- };
681
- const host = region === "global" ? "speech.googleapis.com" : `${region}-speech.googleapis.com`;
682
- const url = `https://${host}/v2/projects/${this.config.project}/locations/${region}/recognizers/_:recognize`;
683
- const {
684
- value: response,
685
- responseHeaders,
686
- rawValue: rawResponse
687
- } = await postJsonToApi3({
688
- url,
689
- headers: combineHeaders3(
690
- this.config.headers ? await resolve3(this.config.headers) : void 0,
691
- options.headers
692
- ),
693
- body: requestBody,
694
- failedResponseHandler: googleVertexFailedResponseHandler,
695
- successfulResponseHandler: createJsonResponseHandler3(
696
- googleVertexTranscriptionResponseSchema
697
- ),
698
- abortSignal: options.abortSignal,
699
- fetch: this.config.fetch
700
- });
701
- const results = response.results ?? [];
702
- const text = results.map((result) => result.alternatives?.[0]?.transcript ?? "").join(" ").trim();
703
- const segments = results.flatMap(
704
- (result) => result.alternatives?.[0]?.words?.flatMap((word) => {
705
- const startSecond = parseDurationSeconds(word.startOffset);
706
- const endSecond = parseDurationSeconds(word.endOffset);
707
- return word.word == null || startSecond == null || endSecond == null ? [] : [{ text: word.word, startSecond, endSecond }];
708
- }) ?? []
709
- );
710
- const language = convertBcp47ToIso6391(results[0]?.languageCode);
711
- return {
712
- text,
713
- segments,
714
- language,
715
- durationInSeconds: parseDurationSeconds(
716
- response.metadata?.totalBilledDuration
717
- ),
718
- warnings,
719
- response: {
720
- timestamp: currentDate,
721
- modelId: this.modelId,
722
- headers: responseHeaders,
723
- body: rawResponse
724
- }
725
- };
726
- }
458
+ var GoogleVertexTranscriptionModel = class GoogleVertexTranscriptionModel {
459
+ static [WORKFLOW_SERIALIZE](model) {
460
+ return serializeModelOptions({
461
+ modelId: model.modelId,
462
+ config: model.config
463
+ });
464
+ }
465
+ static [WORKFLOW_DESERIALIZE](options) {
466
+ return new GoogleVertexTranscriptionModel(options.modelId, options.config);
467
+ }
468
+ get provider() {
469
+ return this.config.provider;
470
+ }
471
+ constructor(modelId, config) {
472
+ this.modelId = modelId;
473
+ this.config = config;
474
+ this.specificationVersion = "v4";
475
+ }
476
+ async doGenerate(options) {
477
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
478
+ const warnings = [];
479
+ let googleOptions;
480
+ for (const provider of [
481
+ "googleVertex",
482
+ "vertex",
483
+ "google"
484
+ ]) {
485
+ googleOptions = await parseProviderOptions({
486
+ provider,
487
+ providerOptions: options.providerOptions,
488
+ schema: googleVertexTranscriptionProviderOptionsSchema
489
+ });
490
+ if (googleOptions != null) break;
491
+ }
492
+ const region = googleOptions?.region ?? this.config.location;
493
+ const languageCodes = googleOptions?.languageCodes ?? ["auto"];
494
+ const content = typeof options.audio === "string" ? options.audio : convertUint8ArrayToBase64(options.audio);
495
+ const requestBody = {
496
+ config: {
497
+ model: this.modelId,
498
+ languageCodes,
499
+ autoDecodingConfig: {},
500
+ features: {
501
+ enableWordTimeOffsets: googleOptions?.enableWordTimeOffsets ?? true,
502
+ enableAutomaticPunctuation: googleOptions?.enableAutomaticPunctuation ?? true
503
+ }
504
+ },
505
+ content
506
+ };
507
+ const url = `https://${region === "global" ? "speech.googleapis.com" : `${region}-speech.googleapis.com`}/v2/projects/${this.config.project}/locations/${region}/recognizers/_:recognize`;
508
+ const { value: response, responseHeaders, rawValue: rawResponse } = await postJsonToApi({
509
+ url,
510
+ headers: combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, options.headers),
511
+ body: requestBody,
512
+ failedResponseHandler: googleVertexFailedResponseHandler,
513
+ successfulResponseHandler: createJsonResponseHandler(googleVertexTranscriptionResponseSchema),
514
+ abortSignal: options.abortSignal,
515
+ fetch: this.config.fetch
516
+ });
517
+ const results = response.results ?? [];
518
+ return {
519
+ text: results.map((result) => result.alternatives?.[0]?.transcript ?? "").join(" ").trim(),
520
+ segments: results.flatMap((result) => result.alternatives?.[0]?.words?.flatMap((word) => {
521
+ const startSecond = parseDurationSeconds(word.startOffset);
522
+ const endSecond = parseDurationSeconds(word.endOffset);
523
+ return word.word == null || startSecond == null || endSecond == null ? [] : [{
524
+ text: word.word,
525
+ startSecond,
526
+ endSecond
527
+ }];
528
+ }) ?? []),
529
+ language: convertBcp47ToIso6391(results[0]?.languageCode),
530
+ durationInSeconds: parseDurationSeconds(response.metadata?.totalBilledDuration),
531
+ warnings,
532
+ response: {
533
+ timestamp: currentDate,
534
+ modelId: this.modelId,
535
+ headers: responseHeaders,
536
+ body: rawResponse
537
+ }
538
+ };
539
+ }
727
540
  };
728
- var googleVertexTranscriptionResponseSchema = z6.object({
729
- results: z6.array(
730
- z6.object({
731
- alternatives: z6.array(
732
- z6.object({
733
- transcript: z6.string().nullish(),
734
- words: z6.array(
735
- z6.object({
736
- word: z6.string().nullish(),
737
- startOffset: z6.string().nullish(),
738
- endOffset: z6.string().nullish()
739
- })
740
- ).nullish()
741
- })
742
- ).nullish(),
743
- languageCode: z6.string().nullish()
744
- })
745
- ).nullish(),
746
- metadata: z6.object({
747
- totalBilledDuration: z6.string().nullish()
748
- }).nullish()
541
+ const googleVertexTranscriptionResponseSchema = z.object({
542
+ results: z.array(z.object({
543
+ alternatives: z.array(z.object({
544
+ transcript: z.string().nullish(),
545
+ words: z.array(z.object({
546
+ word: z.string().nullish(),
547
+ startOffset: z.string().nullish(),
548
+ endOffset: z.string().nullish()
549
+ })).nullish()
550
+ })).nullish(),
551
+ languageCode: z.string().nullish()
552
+ })).nullish(),
553
+ metadata: z.object({ totalBilledDuration: z.string().nullish() }).nullish()
749
554
  });
750
-
751
- // src/gemini-transcription/google-vertex-gemini-transcription-model.ts
752
- import {
753
- InvalidArgumentError
754
- } from "@ai-sdk/provider";
755
- import {
756
- combineHeaders as combineHeaders4,
757
- connectToWebSocket,
758
- convertToBase64 as convertToBase642,
759
- createJsonResponseHandler as createJsonResponseHandler4,
760
- parseProviderOptions as parseProviderOptions3,
761
- postJsonToApi as postJsonToApi4,
762
- resolve as resolve4,
763
- safeParseJSON,
764
- serializeModelOptions as serializeModelOptions5,
765
- waitForWebSocketBufferDrain,
766
- WORKFLOW_DESERIALIZE as WORKFLOW_DESERIALIZE5,
767
- WORKFLOW_SERIALIZE as WORKFLOW_SERIALIZE5
768
- } from "@ai-sdk/provider-utils";
769
- import { z as z8 } from "zod/v4";
770
-
771
- // src/gemini-transcription/google-vertex-gemini-transcription-model-options.ts
772
- import { z as z7 } from "zod/v4";
773
- var googleVertexGeminiTranscriptionModelOptions = z7.object({
774
- /**
775
- * BCP-47 language codes providing hints about the languages present in the
776
- * audio. If omitted or empty, defaults to automatic language detection.
777
- */
778
- languageCodes: z7.array(z7.string()).optional(),
779
- /**
780
- * Custom vocabulary phrases, which bias the speech recognition model
781
- * toward recognizing specific terms.
782
- */
783
- customVocabulary: z7.array(z7.string()).optional(),
784
- /**
785
- * Enables word-level timestamp generation.
786
- */
787
- wordTimestamp: z7.boolean().optional(),
788
- /**
789
- * Enables speaker diarization.
790
- */
791
- diarization: z7.boolean().optional(),
792
- /**
793
- * Transcription output formatting mode.
794
- *
795
- * - `VERBATIM` (default): exact literal transcript preserving filler
796
- * words, repetitions, and false starts.
797
- * - `SMART`: cleans up and structures the transcript in real time —
798
- * disfluency removal, inline self-corrections, structured formatting
799
- * (lists, numbers, dates, paragraph breaks), and grammar/casing polish.
800
- */
801
- mode: z7.enum(["SMART", "VERBATIM"]).optional()
555
+ //#endregion
556
+ //#region src/gemini-transcription/google-vertex-gemini-transcription-model-options.ts
557
+ /**
558
+ * Speech recognition options for Gemini transcription models on Vertex,
559
+ * shared by unary (`gemini-3.5-transcribe`) and live
560
+ * (`gemini-3.5-transcribe-live`) variants. Maps onto Google's
561
+ * `AudioTranscriptionConfig`.
562
+ */
563
+ const googleVertexGeminiTranscriptionModelOptions = z.object({
564
+ /**
565
+ * BCP-47 language codes providing hints about the languages present in the
566
+ * audio. If omitted or empty, defaults to automatic language detection.
567
+ */
568
+ languageCodes: z.array(z.string()).optional(),
569
+ /**
570
+ * Custom vocabulary phrases, which bias the speech recognition model
571
+ * toward recognizing specific terms.
572
+ */
573
+ customVocabulary: z.array(z.string()).optional(),
574
+ /**
575
+ * Enables word-level timestamp generation.
576
+ */
577
+ wordTimestamp: z.boolean().optional(),
578
+ /**
579
+ * Enables speaker diarization.
580
+ */
581
+ diarization: z.boolean().optional(),
582
+ /**
583
+ * Transcription output formatting mode.
584
+ *
585
+ * - `VERBATIM` (default): exact literal transcript preserving filler
586
+ * words, repetitions, and false starts.
587
+ * - `SMART`: cleans up and structures the transcript in real time —
588
+ * disfluency removal, inline self-corrections, structured formatting
589
+ * (lists, numbers, dates, paragraph breaks), and grammar/casing polish.
590
+ */
591
+ mode: z.enum(["SMART", "VERBATIM"]).optional()
802
592
  });
803
-
804
- // src/gemini-transcription/google-vertex-gemini-transcription-model.ts
805
- var liveWebSocketPath = "google.cloud.aiplatform.v1.LlmBidiService/BidiGenerateContent";
806
- var defaultFinishGraceMs = 3e3;
593
+ //#endregion
594
+ //#region src/gemini-transcription/google-vertex-gemini-transcription-model.ts
595
+ const liveWebSocketPath = "google.cloud.aiplatform.v1.LlmBidiService/BidiGenerateContent";
596
+ /**
597
+ * After the input audio has ended, finish when no terminal signal
598
+ * (`turnComplete` / idle `interactionStatus`) arrives within this window.
599
+ * Trailing transcripts reset the timer.
600
+ */
601
+ const defaultFinishGraceMs = 3e3;
602
+ /** Live transcription is only supported by `*-live` model variants. */
807
603
  function isLiveTranscriptionModelId(modelId) {
808
- return modelId.includes("-live");
604
+ return modelId.includes("-live");
809
605
  }
606
+ /** Regional Vertex hostname (mirrors the provider's base-URL host rules). */
810
607
  function vertexHost(location) {
811
- if (location === "global") return "aiplatform.googleapis.com";
812
- if (location === "eu" || location === "us") {
813
- return `aiplatform.${location}.rep.googleapis.com`;
814
- }
815
- return `${location}-aiplatform.googleapis.com`;
608
+ if (location === "global") return "aiplatform.googleapis.com";
609
+ if (location === "eu" || location === "us") return `aiplatform.${location}.rep.googleapis.com`;
610
+ return `${location}-aiplatform.googleapis.com`;
816
611
  }
817
- var GoogleVertexGeminiTranscriptionModel = class _GoogleVertexGeminiTranscriptionModel {
818
- constructor(modelId, config) {
819
- this.modelId = modelId;
820
- this.config = config;
821
- this.specificationVersion = "v4";
822
- }
823
- static [WORKFLOW_SERIALIZE5](model) {
824
- return serializeModelOptions5({
825
- modelId: model.modelId,
826
- config: model.config
827
- });
828
- }
829
- static [WORKFLOW_DESERIALIZE5](options) {
830
- return new _GoogleVertexGeminiTranscriptionModel(
831
- options.modelId,
832
- options.config
833
- );
834
- }
835
- get provider() {
836
- return this.config.provider;
837
- }
838
- async parseOptions(providerOptions) {
839
- for (const provider of ["googleVertex", "vertex", "google"]) {
840
- const parsed = await parseProviderOptions3({
841
- provider,
842
- providerOptions,
843
- schema: googleVertexGeminiTranscriptionModelOptions
844
- });
845
- if (parsed != null) return parsed;
846
- }
847
- }
848
- async doGenerate(options) {
849
- if (isLiveTranscriptionModelId(this.modelId)) {
850
- throw new InvalidArgumentError({
851
- argument: "modelId",
852
- message: `Model '${this.modelId}' only supports streaming transcription. Use experimental_streamTranscribe, or a unary model such as 'gemini-3.5-transcribe'.`
853
- });
854
- }
855
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
856
- const warnings = [];
857
- const googleOptions = await this.parseOptions(options.providerOptions);
858
- const audioTranscriptionConfig = buildAudioTranscriptionConfig(googleOptions);
859
- const requestBody = {
860
- contents: [
861
- {
862
- role: "user",
863
- parts: [
864
- {
865
- inlineData: {
866
- mimeType: options.mediaType,
867
- data: convertToBase642(options.audio)
868
- }
869
- }
870
- ]
871
- }
872
- ],
873
- ...audioTranscriptionConfig != null ? { generationConfig: { audioTranscriptionConfig } } : {}
874
- };
875
- const {
876
- value: response,
877
- responseHeaders,
878
- rawValue: rawResponse
879
- } = await postJsonToApi4({
880
- url: `${this.config.baseURL}/models/${this.modelId}:generateContent`,
881
- headers: combineHeaders4(
882
- this.config.headers ? await resolve4(this.config.headers) : void 0,
883
- options.headers
884
- ),
885
- body: requestBody,
886
- failedResponseHandler: googleVertexFailedResponseHandler,
887
- successfulResponseHandler: createJsonResponseHandler4(
888
- googleVertexGeminiTranscriptionResponseSchema
889
- ),
890
- abortSignal: options.abortSignal,
891
- fetch: this.config.fetch
892
- });
893
- const parts = response.candidates?.[0]?.content?.parts ?? [];
894
- const plainText = parts.map((part) => part.text ?? "").join("");
895
- const transcriptionText = parts.map((part) => part.audioTranscription?.text ?? "").join("");
896
- const text = plainText !== "" ? plainText : transcriptionText;
897
- let language;
898
- const segments = [];
899
- for (const part of parts) {
900
- const transcription = part.audioTranscription;
901
- if (transcription == null) continue;
902
- language ??= transcription.languageCode ?? void 0;
903
- for (const word of transcription.words ?? []) {
904
- const startSecond = parseOffsetSeconds(word.startOffset);
905
- const endSecond = parseOffsetSeconds(word.endOffset);
906
- if (word.word == null || startSecond == null || endSecond == null) {
907
- continue;
908
- }
909
- segments.push({ text: word.word, startSecond, endSecond });
910
- }
911
- }
912
- return {
913
- text,
914
- segments,
915
- language,
916
- durationInSeconds: void 0,
917
- warnings,
918
- response: {
919
- timestamp: currentDate,
920
- modelId: this.modelId,
921
- headers: responseHeaders,
922
- body: rawResponse
923
- },
924
- ...response.usageMetadata != null ? {
925
- providerMetadata: {
926
- google: { usageMetadata: response.usageMetadata }
927
- }
928
- } : {}
929
- };
930
- }
931
- async doStream(options) {
932
- if (!isLiveTranscriptionModelId(this.modelId)) {
933
- throw new InvalidArgumentError({
934
- argument: "modelId",
935
- message: `Model '${this.modelId}' does not support streaming transcription. Use a live model such as 'gemini-3.5-transcribe-live'.`
936
- });
937
- }
938
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
939
- const warnings = [];
940
- const googleOptions = await this.parseOptions(options.providerOptions);
941
- validateLiveInputAudioFormat(options.inputAudioFormat);
942
- const headers = combineHeaders4(
943
- this.config.headers ? await resolve4(this.config.headers) : void 0,
944
- options.headers
945
- );
946
- const { project, location } = this.config;
947
- const modelResource = `projects/${project}/locations/${location}/publishers/google/models/${this.modelId}`;
948
- const url = new URL(
949
- `wss://${vertexHost(location)}/ws/${liveWebSocketPath}`
950
- );
951
- const setup = {
952
- model: modelResource,
953
- inputAudioTranscription: buildAudioTranscriptionConfig(googleOptions) ?? {}
954
- };
955
- return {
956
- request: { body: setup },
957
- response: {
958
- timestamp: currentDate,
959
- modelId: this.modelId
960
- },
961
- stream: createVertexLiveTranscriptionStream({
962
- webSocket: this.config.webSocket,
963
- url,
964
- headers,
965
- setup,
966
- inputAudioRate: options.inputAudioFormat.rate ?? 16e3,
967
- finishGraceMs: this.config._internal?.finishGraceMs ?? defaultFinishGraceMs,
968
- warnings,
969
- audio: options.audio,
970
- abortSignal: options.abortSignal,
971
- includeRawChunks: options.includeRawChunks
972
- })
973
- };
974
- }
612
+ /**
613
+ * Gemini transcription on Vertex AI. Unary variants transcribe via
614
+ * `generateContent`; live variants stream over the Vertex Live API WebSocket
615
+ * (`LlmBidiService/BidiGenerateContent`) with OAuth Bearer authentication
616
+ * from the provider's resolved headers.
617
+ */
618
+ var GoogleVertexGeminiTranscriptionModel = class GoogleVertexGeminiTranscriptionModel {
619
+ static [WORKFLOW_SERIALIZE](model) {
620
+ return serializeModelOptions({
621
+ modelId: model.modelId,
622
+ config: model.config
623
+ });
624
+ }
625
+ static [WORKFLOW_DESERIALIZE](options) {
626
+ return new GoogleVertexGeminiTranscriptionModel(options.modelId, options.config);
627
+ }
628
+ get provider() {
629
+ return this.config.provider;
630
+ }
631
+ constructor(modelId, config) {
632
+ this.modelId = modelId;
633
+ this.config = config;
634
+ this.specificationVersion = "v4";
635
+ }
636
+ async parseOptions(providerOptions) {
637
+ for (const provider of [
638
+ "googleVertex",
639
+ "vertex",
640
+ "google"
641
+ ]) {
642
+ const parsed = await parseProviderOptions({
643
+ provider,
644
+ providerOptions,
645
+ schema: googleVertexGeminiTranscriptionModelOptions
646
+ });
647
+ if (parsed != null) return parsed;
648
+ }
649
+ }
650
+ async doGenerate(options) {
651
+ if (isLiveTranscriptionModelId(this.modelId)) throw new InvalidArgumentError({
652
+ argument: "modelId",
653
+ message: `Model '${this.modelId}' only supports streaming transcription. Use experimental_streamTranscribe, or a unary model such as 'gemini-3.5-transcribe'.`
654
+ });
655
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
656
+ const warnings = [];
657
+ const audioTranscriptionConfig = buildAudioTranscriptionConfig(await this.parseOptions(options.providerOptions));
658
+ const requestBody = {
659
+ contents: [{
660
+ role: "user",
661
+ parts: [{ inlineData: {
662
+ mimeType: options.mediaType,
663
+ data: convertToBase64(options.audio)
664
+ } }]
665
+ }],
666
+ ...audioTranscriptionConfig != null ? { generationConfig: { audioTranscriptionConfig } } : {}
667
+ };
668
+ const { value: response, responseHeaders, rawValue: rawResponse } = await postJsonToApi({
669
+ url: `${this.config.baseURL}/models/${this.modelId}:generateContent`,
670
+ headers: combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, options.headers),
671
+ body: requestBody,
672
+ failedResponseHandler: googleVertexFailedResponseHandler,
673
+ successfulResponseHandler: createJsonResponseHandler(googleVertexGeminiTranscriptionResponseSchema),
674
+ abortSignal: options.abortSignal,
675
+ fetch: this.config.fetch
676
+ });
677
+ const parts = response.candidates?.[0]?.content?.parts ?? [];
678
+ const plainText = parts.map((part) => part.text ?? "").join("");
679
+ const transcriptionText = parts.map((part) => part.audioTranscription?.text ?? "").join("");
680
+ const text = plainText !== "" ? plainText : transcriptionText;
681
+ let language;
682
+ const segments = [];
683
+ for (const part of parts) {
684
+ const transcription = part.audioTranscription;
685
+ if (transcription == null) continue;
686
+ language ??= transcription.languageCode ?? void 0;
687
+ for (const word of transcription.words ?? []) {
688
+ const startSecond = parseOffsetSeconds(word.startOffset);
689
+ const endSecond = parseOffsetSeconds(word.endOffset);
690
+ if (word.word == null || startSecond == null || endSecond == null) continue;
691
+ segments.push({
692
+ text: word.word,
693
+ startSecond,
694
+ endSecond
695
+ });
696
+ }
697
+ }
698
+ return {
699
+ text,
700
+ segments,
701
+ language,
702
+ durationInSeconds: void 0,
703
+ warnings,
704
+ response: {
705
+ timestamp: currentDate,
706
+ modelId: this.modelId,
707
+ headers: responseHeaders,
708
+ body: rawResponse
709
+ },
710
+ ...response.usageMetadata != null ? { providerMetadata: { google: { usageMetadata: response.usageMetadata } } } : {}
711
+ };
712
+ }
713
+ async doStream(options) {
714
+ if (!isLiveTranscriptionModelId(this.modelId)) throw new InvalidArgumentError({
715
+ argument: "modelId",
716
+ message: `Model '${this.modelId}' does not support streaming transcription. Use a live model such as 'gemini-3.5-transcribe-live'.`
717
+ });
718
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
719
+ const warnings = [];
720
+ const googleOptions = await this.parseOptions(options.providerOptions);
721
+ validateLiveInputAudioFormat(options.inputAudioFormat);
722
+ const headers = combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, options.headers);
723
+ const { project, location } = this.config;
724
+ const modelResource = `projects/${project}/locations/${location}/publishers/google/models/${this.modelId}`;
725
+ const url = new URL(`wss://${vertexHost(location)}/ws/${liveWebSocketPath}`);
726
+ const setup = {
727
+ model: modelResource,
728
+ inputAudioTranscription: buildAudioTranscriptionConfig(googleOptions) ?? {}
729
+ };
730
+ return {
731
+ request: { body: setup },
732
+ response: {
733
+ timestamp: currentDate,
734
+ modelId: this.modelId
735
+ },
736
+ stream: createVertexLiveTranscriptionStream({
737
+ webSocket: this.config.webSocket,
738
+ url,
739
+ headers,
740
+ setup,
741
+ inputAudioRate: options.inputAudioFormat.rate ?? 16e3,
742
+ finishGraceMs: this.config._internal?.finishGraceMs ?? defaultFinishGraceMs,
743
+ warnings,
744
+ audio: options.audio,
745
+ abortSignal: options.abortSignal,
746
+ includeRawChunks: options.includeRawChunks
747
+ })
748
+ };
749
+ }
975
750
  };
976
- function createVertexLiveTranscriptionStream({
977
- webSocket,
978
- url,
979
- headers,
980
- setup,
981
- inputAudioRate,
982
- finishGraceMs,
983
- warnings,
984
- audio,
985
- abortSignal,
986
- includeRawChunks
987
- }) {
988
- let finished = false;
989
- let cleanup = () => {
990
- };
991
- return new ReadableStream({
992
- start: (controller) => {
993
- let audioReader;
994
- let connection;
995
- let resolveSetupComplete;
996
- const setupComplete = new Promise((resolvePromise) => {
997
- resolveSetupComplete = resolvePromise;
998
- });
999
- let segmentCounter = 0;
1000
- let segmentBuffer = "";
1001
- let fullText = "";
1002
- let latestInterim = "";
1003
- let language;
1004
- let audioEnded = false;
1005
- let usageMetadata;
1006
- let finishTimer;
1007
- const segmentId = () => `google-segment-${segmentCounter}`;
1008
- const cancelPendingFinish = () => {
1009
- if (finishTimer != null) {
1010
- clearTimeout(finishTimer);
1011
- finishTimer = void 0;
1012
- }
1013
- };
1014
- const schedulePendingFinish = () => {
1015
- if (finished || !audioEnded) return;
1016
- cancelPendingFinish();
1017
- finishTimer = setTimeout(() => {
1018
- finishTimer = void 0;
1019
- finish();
1020
- }, finishGraceMs);
1021
- };
1022
- cleanup = (closeCode) => {
1023
- cancelPendingFinish();
1024
- if (audioReader != null) {
1025
- void audioReader.cancel().catch(() => {
1026
- });
1027
- } else {
1028
- void audio.cancel().catch(() => {
1029
- });
1030
- }
1031
- connection?.close(closeCode);
1032
- };
1033
- const finishWithError = (error) => {
1034
- if (finished) return;
1035
- finished = true;
1036
- cleanup();
1037
- controller.error(error);
1038
- };
1039
- const completeSegment = () => {
1040
- if (segmentBuffer === "") {
1041
- if (latestInterim === "") return;
1042
- segmentBuffer = latestInterim;
1043
- }
1044
- latestInterim = "";
1045
- controller.enqueue({
1046
- type: "transcript-final",
1047
- id: segmentId(),
1048
- text: segmentBuffer
1049
- });
1050
- fullText += fullText === "" ? segmentBuffer : ` ${segmentBuffer}`;
1051
- segmentBuffer = "";
1052
- segmentCounter++;
1053
- };
1054
- const finish = () => {
1055
- if (finished) return;
1056
- completeSegment();
1057
- finished = true;
1058
- controller.enqueue({
1059
- type: "finish",
1060
- text: fullText,
1061
- segments: [],
1062
- language,
1063
- durationInSeconds: void 0,
1064
- ...usageMetadata != null ? { providerMetadata: { google: { usageMetadata } } } : {}
1065
- });
1066
- controller.close();
1067
- cleanup(1e3);
1068
- };
1069
- const sendAudio = async (socket) => {
1070
- audioReader = audio.getReader();
1071
- try {
1072
- while (true) {
1073
- const { done, value } = await audioReader.read();
1074
- if (done || finished) break;
1075
- socket.send(
1076
- JSON.stringify({
1077
- realtimeInput: {
1078
- audio: {
1079
- data: convertToBase642(value),
1080
- mimeType: `audio/pcm;rate=${inputAudioRate}`
1081
- }
1082
- }
1083
- })
1084
- );
1085
- await waitForWebSocketBufferDrain(socket);
1086
- }
1087
- } finally {
1088
- audioReader.releaseLock();
1089
- audioReader = void 0;
1090
- }
1091
- if (!finished) {
1092
- socket.send(
1093
- JSON.stringify({ realtimeInput: { audioStreamEnd: true } })
1094
- );
1095
- audioEnded = true;
1096
- schedulePendingFinish();
1097
- }
1098
- };
1099
- connection = connectToWebSocket({
1100
- url,
1101
- headers,
1102
- webSocket,
1103
- abortSignal,
1104
- onAbort: finishWithError,
1105
- onProcessingError: finishWithError,
1106
- onOpen: (socket) => {
1107
- controller.enqueue({ type: "stream-start", warnings });
1108
- socket.send(JSON.stringify({ setup }));
1109
- void setupComplete.then(() => finished ? void 0 : sendAudio(socket)).catch(finishWithError);
1110
- },
1111
- onMessageText: async (text) => {
1112
- if (finished) return;
1113
- const parsed = await safeParseJSON({ text });
1114
- if (!parsed.success) return;
1115
- const message = parsed.value;
1116
- if (includeRawChunks) {
1117
- controller.enqueue({ type: "raw", rawValue: message });
1118
- }
1119
- if (message.setupComplete != null) {
1120
- resolveSetupComplete();
1121
- }
1122
- if (message.usageMetadata != null) {
1123
- usageMetadata = message.usageMetadata;
1124
- }
1125
- if (message.error != null) {
1126
- finishWithError(
1127
- new Error(message.error.message ?? "Vertex Live API error")
1128
- );
1129
- return;
1130
- }
1131
- const serverContent = message.serverContent;
1132
- const interim = serverContent?.interimInputTranscription;
1133
- if (interim?.text) {
1134
- schedulePendingFinish();
1135
- latestInterim = interim.text;
1136
- controller.enqueue({
1137
- type: "transcript-partial",
1138
- id: segmentId(),
1139
- text: interim.text
1140
- });
1141
- }
1142
- const transcription = serverContent?.inputTranscription ?? message.inputTranscription;
1143
- if (transcription != null) {
1144
- if (transcription.languageCode != null) {
1145
- language = transcription.languageCode;
1146
- }
1147
- if (transcription.text) {
1148
- schedulePendingFinish();
1149
- latestInterim = "";
1150
- segmentBuffer += transcription.text;
1151
- controller.enqueue({
1152
- type: "transcript-delta",
1153
- id: segmentId(),
1154
- delta: transcription.text
1155
- });
1156
- }
1157
- if (transcription.finished === true) {
1158
- completeSegment();
1159
- }
1160
- }
1161
- if (serverContent?.turnComplete) {
1162
- completeSegment();
1163
- }
1164
- const interactionStatus = serverContent?.interactionStatus;
1165
- if (audioEnded && (interactionStatus === "IDLE" || interactionStatus === "REQUIRES_ACTION" || serverContent?.turnComplete === true && interactionStatus == null)) {
1166
- finish();
1167
- }
1168
- },
1169
- onSocketError: () => {
1170
- finishWithError(
1171
- new Error(
1172
- "Vertex Live transcription error." + (webSocket == null ? " Note: the native WebSocket implementation cannot send the Authorization header required by Vertex. Pass a header-capable WebSocket implementation (e.g. the 'ws' package) via createVertex({ webSocket })." : "")
1173
- )
1174
- );
1175
- },
1176
- onClose: ({ code, reason }) => {
1177
- if (finished) return;
1178
- if (audioEnded) {
1179
- finish();
1180
- return;
1181
- }
1182
- finishWithError(
1183
- new Error(
1184
- `Vertex Live transcription WebSocket closed unexpectedly before finishing (code ${code ?? "unknown"}${reason ? `, reason: ${reason}` : ""}).`
1185
- )
1186
- );
1187
- }
1188
- });
1189
- },
1190
- cancel: () => {
1191
- if (finished) return;
1192
- finished = true;
1193
- cleanup();
1194
- }
1195
- });
751
+ function createVertexLiveTranscriptionStream({ webSocket, url, headers, setup, inputAudioRate, finishGraceMs, warnings, audio, abortSignal, includeRawChunks }) {
752
+ let finished = false;
753
+ let cleanup = () => {};
754
+ return new ReadableStream({
755
+ start: (controller) => {
756
+ let audioReader;
757
+ let connection;
758
+ let resolveSetupComplete;
759
+ const setupComplete = new Promise((resolvePromise) => {
760
+ resolveSetupComplete = resolvePromise;
761
+ });
762
+ let segmentCounter = 0;
763
+ let segmentBuffer = "";
764
+ let fullText = "";
765
+ let latestInterim = "";
766
+ let language;
767
+ let audioEnded = false;
768
+ let usageMetadata;
769
+ let finishTimer;
770
+ const segmentId = () => `google-segment-${segmentCounter}`;
771
+ const cancelPendingFinish = () => {
772
+ if (finishTimer != null) {
773
+ clearTimeout(finishTimer);
774
+ finishTimer = void 0;
775
+ }
776
+ };
777
+ const schedulePendingFinish = () => {
778
+ if (finished || !audioEnded) return;
779
+ cancelPendingFinish();
780
+ finishTimer = setTimeout(() => {
781
+ finishTimer = void 0;
782
+ finish();
783
+ }, finishGraceMs);
784
+ };
785
+ cleanup = (closeCode) => {
786
+ cancelPendingFinish();
787
+ if (audioReader != null) audioReader.cancel().catch(() => {});
788
+ else audio.cancel().catch(() => {});
789
+ connection?.close(closeCode);
790
+ };
791
+ const finishWithError = (error) => {
792
+ if (finished) return;
793
+ finished = true;
794
+ cleanup();
795
+ controller.error(error);
796
+ };
797
+ const completeSegment = () => {
798
+ if (segmentBuffer === "") {
799
+ if (latestInterim === "") return;
800
+ segmentBuffer = latestInterim;
801
+ }
802
+ latestInterim = "";
803
+ controller.enqueue({
804
+ type: "transcript-final",
805
+ id: segmentId(),
806
+ text: segmentBuffer
807
+ });
808
+ fullText += fullText === "" ? segmentBuffer : ` ${segmentBuffer}`;
809
+ segmentBuffer = "";
810
+ segmentCounter++;
811
+ };
812
+ const finish = () => {
813
+ if (finished) return;
814
+ completeSegment();
815
+ finished = true;
816
+ controller.enqueue({
817
+ type: "finish",
818
+ text: fullText,
819
+ segments: [],
820
+ language,
821
+ durationInSeconds: void 0,
822
+ ...usageMetadata != null ? { providerMetadata: { google: { usageMetadata } } } : {}
823
+ });
824
+ controller.close();
825
+ cleanup(1e3);
826
+ };
827
+ const sendAudio = async (socket) => {
828
+ audioReader = audio.getReader();
829
+ try {
830
+ while (true) {
831
+ const { done, value } = await audioReader.read();
832
+ if (done || finished) break;
833
+ socket.send(JSON.stringify({ realtimeInput: { audio: {
834
+ data: convertToBase64(value),
835
+ mimeType: `audio/pcm;rate=${inputAudioRate}`
836
+ } } }));
837
+ await waitForWebSocketBufferDrain(socket);
838
+ }
839
+ } finally {
840
+ audioReader.releaseLock();
841
+ audioReader = void 0;
842
+ }
843
+ if (!finished) {
844
+ socket.send(JSON.stringify({ realtimeInput: { audioStreamEnd: true } }));
845
+ audioEnded = true;
846
+ schedulePendingFinish();
847
+ }
848
+ };
849
+ connection = connectToWebSocket({
850
+ url,
851
+ headers,
852
+ webSocket,
853
+ abortSignal,
854
+ onAbort: finishWithError,
855
+ onProcessingError: finishWithError,
856
+ onOpen: (socket) => {
857
+ controller.enqueue({
858
+ type: "stream-start",
859
+ warnings
860
+ });
861
+ socket.send(JSON.stringify({ setup }));
862
+ setupComplete.then(() => finished ? void 0 : sendAudio(socket)).catch(finishWithError);
863
+ },
864
+ onMessageText: async (text) => {
865
+ if (finished) return;
866
+ const parsed = await safeParseJSON({ text });
867
+ if (!parsed.success) return;
868
+ const message = parsed.value;
869
+ if (includeRawChunks) controller.enqueue({
870
+ type: "raw",
871
+ rawValue: message
872
+ });
873
+ if (message.setupComplete != null) resolveSetupComplete();
874
+ if (message.usageMetadata != null) usageMetadata = message.usageMetadata;
875
+ if (message.error != null) {
876
+ finishWithError(new Error(message.error.message ?? "Vertex Live API error"));
877
+ return;
878
+ }
879
+ const serverContent = message.serverContent;
880
+ const interim = serverContent?.interimInputTranscription;
881
+ if (interim?.text) {
882
+ schedulePendingFinish();
883
+ latestInterim = interim.text;
884
+ controller.enqueue({
885
+ type: "transcript-partial",
886
+ id: segmentId(),
887
+ text: interim.text
888
+ });
889
+ }
890
+ const transcription = serverContent?.inputTranscription ?? message.inputTranscription;
891
+ if (transcription != null) {
892
+ if (transcription.languageCode != null) language = transcription.languageCode;
893
+ if (transcription.text) {
894
+ schedulePendingFinish();
895
+ latestInterim = "";
896
+ segmentBuffer += transcription.text;
897
+ controller.enqueue({
898
+ type: "transcript-delta",
899
+ id: segmentId(),
900
+ delta: transcription.text
901
+ });
902
+ }
903
+ if (transcription.finished === true) completeSegment();
904
+ }
905
+ if (serverContent?.turnComplete) completeSegment();
906
+ const interactionStatus = serverContent?.interactionStatus;
907
+ if (audioEnded && (interactionStatus === "IDLE" || interactionStatus === "REQUIRES_ACTION" || serverContent?.turnComplete === true && interactionStatus == null)) finish();
908
+ },
909
+ onSocketError: () => {
910
+ finishWithError(/* @__PURE__ */ new Error("Vertex Live transcription error." + (webSocket == null ? " Note: the native WebSocket implementation cannot send the Authorization header required by Vertex. Pass a header-capable WebSocket implementation (e.g. the 'ws' package) via createVertex({ webSocket })." : "")));
911
+ },
912
+ onClose: ({ code, reason }) => {
913
+ if (finished) return;
914
+ if (audioEnded) {
915
+ finish();
916
+ return;
917
+ }
918
+ finishWithError(/* @__PURE__ */ new Error(`Vertex Live transcription WebSocket closed unexpectedly before finishing (code ${code ?? "unknown"}${reason ? `, reason: ${reason}` : ""}).`));
919
+ }
920
+ });
921
+ },
922
+ cancel: () => {
923
+ if (finished) return;
924
+ finished = true;
925
+ cleanup();
926
+ }
927
+ });
1196
928
  }
929
+ /**
930
+ * Builds Google's `AudioTranscriptionConfig` from provider options; returns
931
+ * undefined when no options are set.
932
+ */
1197
933
  function buildAudioTranscriptionConfig(options) {
1198
- if (options == null) return void 0;
1199
- const config = {};
1200
- if (options.languageCodes != null) {
1201
- config.languageCodes = options.languageCodes;
1202
- }
1203
- if (options.customVocabulary != null) {
1204
- config.customVocabulary = options.customVocabulary;
1205
- }
1206
- if (options.wordTimestamp != null) {
1207
- config.wordTimestamp = options.wordTimestamp;
1208
- }
1209
- if (options.diarization != null) {
1210
- config.diarization = options.diarization;
1211
- }
1212
- if (options.mode != null) {
1213
- config.mode = options.mode;
1214
- }
1215
- return Object.keys(config).length > 0 ? config : void 0;
934
+ if (options == null) return void 0;
935
+ const config = {};
936
+ if (options.languageCodes != null) config.languageCodes = options.languageCodes;
937
+ if (options.customVocabulary != null) config.customVocabulary = options.customVocabulary;
938
+ if (options.wordTimestamp != null) config.wordTimestamp = options.wordTimestamp;
939
+ if (options.diarization != null) config.diarization = options.diarization;
940
+ if (options.mode != null) config.mode = options.mode;
941
+ return Object.keys(config).length > 0 ? config : void 0;
1216
942
  }
1217
943
  function validateLiveInputAudioFormat(inputAudioFormat) {
1218
- if (inputAudioFormat.type !== "audio/pcm" || inputAudioFormat.rate != null && inputAudioFormat.rate !== 16e3) {
1219
- throw new InvalidArgumentError({
1220
- argument: "inputAudioFormat",
1221
- message: "The Gemini Live transcription API only supports 16kHz 16-bit PCM input audio."
1222
- });
1223
- }
944
+ if (inputAudioFormat.type !== "audio/pcm" || inputAudioFormat.rate != null && inputAudioFormat.rate !== 16e3) throw new InvalidArgumentError({
945
+ argument: "inputAudioFormat",
946
+ message: "The Gemini Live transcription API only supports 16kHz 16-bit PCM input audio."
947
+ });
1224
948
  }
949
+ /** Parses a Google duration offset such as `"1s"` or `"9.400s"` to seconds. */
1225
950
  function parseOffsetSeconds(offset) {
1226
- if (offset == null) return void 0;
1227
- const parsed = Number.parseFloat(offset);
1228
- return Number.isFinite(parsed) ? parsed : void 0;
951
+ if (offset == null) return void 0;
952
+ const parsed = Number.parseFloat(offset);
953
+ return Number.isFinite(parsed) ? parsed : void 0;
1229
954
  }
1230
- var googleVertexGeminiTranscriptionWordSchema = z8.object({
1231
- word: z8.string().nullish(),
1232
- startOffset: z8.string().nullish(),
1233
- endOffset: z8.string().nullish()
955
+ const googleVertexGeminiTranscriptionWordSchema = z.object({
956
+ word: z.string().nullish(),
957
+ startOffset: z.string().nullish(),
958
+ endOffset: z.string().nullish()
1234
959
  });
1235
- var googleVertexGeminiTranscriptionResponseSchema = z8.object({
1236
- candidates: z8.array(
1237
- z8.object({
1238
- content: z8.object({
1239
- parts: z8.array(
1240
- z8.object({
1241
- text: z8.string().nullish(),
1242
- audioTranscription: z8.object({
1243
- text: z8.string().nullish(),
1244
- languageCode: z8.string().nullish(),
1245
- speakerLabel: z8.string().nullish(),
1246
- words: z8.array(googleVertexGeminiTranscriptionWordSchema).nullish()
1247
- }).nullish()
1248
- })
1249
- ).nullish()
1250
- }).nullish()
1251
- })
1252
- ).nullish(),
1253
- usageMetadata: z8.record(z8.string(), z8.unknown()).nullish()
960
+ const googleVertexGeminiTranscriptionResponseSchema = z.object({
961
+ candidates: z.array(z.object({ content: z.object({ parts: z.array(z.object({
962
+ text: z.string().nullish(),
963
+ audioTranscription: z.object({
964
+ text: z.string().nullish(),
965
+ languageCode: z.string().nullish(),
966
+ speakerLabel: z.string().nullish(),
967
+ words: z.array(googleVertexGeminiTranscriptionWordSchema).nullish()
968
+ }).nullish()
969
+ })).nullish() }).nullish() })).nullish(),
970
+ usageMetadata: z.record(z.string(), z.unknown()).nullish()
1254
971
  });
1255
-
1256
- // src/google-vertex-video-model.ts
1257
- import {
1258
- AISDKError
1259
- } from "@ai-sdk/provider";
1260
- import {
1261
- combineHeaders as combineHeaders5,
1262
- convertUint8ArrayToBase64 as convertUint8ArrayToBase642,
1263
- createJsonResponseHandler as createJsonResponseHandler5,
1264
- parseProviderOptions as parseProviderOptions4,
1265
- postJsonToApi as postJsonToApi5,
1266
- resolve as resolve5
1267
- } from "@ai-sdk/provider-utils";
1268
- import { z as z10 } from "zod/v4";
1269
-
1270
- // src/google-vertex-video-model-options.ts
1271
- import { lazySchema, zodSchema } from "@ai-sdk/provider-utils";
1272
- import { z as z9 } from "zod/v4";
1273
- var googleVertexVideoModelOptionsSchema = lazySchema(
1274
- () => zodSchema(
1275
- z9.looseObject({
1276
- pollIntervalMs: z9.number().positive().nullish(),
1277
- pollTimeoutMs: z9.number().positive().nullish(),
1278
- personGeneration: z9.enum(["dont_allow", "allow_adult", "allow_all"]).nullish(),
1279
- negativePrompt: z9.string().nullish(),
1280
- generateAudio: z9.boolean().nullish(),
1281
- gcsOutputDirectory: z9.string().nullish(),
1282
- referenceImages: z9.array(
1283
- z9.object({
1284
- bytesBase64Encoded: z9.string().nullish(),
1285
- gcsUri: z9.string().nullish()
1286
- })
1287
- ).nullish()
1288
- })
1289
- )
1290
- );
1291
-
1292
- // src/google-vertex-video-model.ts
972
+ //#endregion
973
+ //#region src/google-vertex-video-model-options.ts
974
+ const googleVertexVideoModelOptionsSchema = lazySchema(() => zodSchema(z.looseObject({
975
+ pollIntervalMs: z.number().positive().nullish(),
976
+ pollTimeoutMs: z.number().positive().nullish(),
977
+ personGeneration: z.enum([
978
+ "dont_allow",
979
+ "allow_adult",
980
+ "allow_all"
981
+ ]).nullish(),
982
+ negativePrompt: z.string().nullish(),
983
+ generateAudio: z.boolean().nullish(),
984
+ gcsOutputDirectory: z.string().nullish(),
985
+ referenceImages: z.array(z.object({
986
+ bytesBase64Encoded: z.string().nullish(),
987
+ gcsUri: z.string().nullish()
988
+ })).nullish()
989
+ })));
990
+ //#endregion
991
+ //#region src/google-vertex-video-model.ts
1293
992
  function getFirstFrameImage(options) {
1294
- return options.frameImages?.find((frame) => frame.frameType === "first_frame")?.image;
993
+ return options.frameImages?.find((frame) => frame.frameType === "first_frame")?.image;
1295
994
  }
1296
995
  function resolveStartImage(options) {
1297
- return getFirstFrameImage(options) ?? options.image;
996
+ return getFirstFrameImage(options) ?? options.image;
1298
997
  }
1299
998
  function getLastFrameImage(options) {
1300
- return options.frameImages?.find((frame) => frame.frameType === "last_frame")?.image;
999
+ return options.frameImages?.find((frame) => frame.frameType === "last_frame")?.image;
1301
1000
  }
1302
1001
  function getInputReferences(options) {
1303
- if (options.frameImages != null && options.frameImages.length > 0) {
1304
- return void 0;
1305
- }
1306
- return options.inputReferences != null && options.inputReferences.length > 0 ? options.inputReferences : void 0;
1002
+ if (options.frameImages != null && options.frameImages.length > 0) return;
1003
+ return options.inputReferences != null && options.inputReferences.length > 0 ? options.inputReferences : void 0;
1307
1004
  }
1308
1005
  function convertFileToVertexImage(file, warnings) {
1309
- if (file.type === "url") {
1310
- if (file.url.startsWith("gs://")) {
1311
- return {
1312
- gcsUri: file.url,
1313
- mimeType: "image/png"
1314
- };
1315
- }
1316
- warnings.push({
1317
- type: "unsupported",
1318
- feature: "URL-based image input",
1319
- details: "Vertex AI video models require base64-encoded images or GCS URIs. URL will be ignored."
1320
- });
1321
- return void 0;
1322
- }
1323
- const base64Data = typeof file.data === "string" ? file.data : convertUint8ArrayToBase642(file.data);
1324
- return {
1325
- bytesBase64Encoded: base64Data,
1326
- mimeType: file.mediaType || "image/png"
1327
- };
1006
+ if (file.type === "url") {
1007
+ if (file.url.startsWith("gs://")) return {
1008
+ gcsUri: file.url,
1009
+ mimeType: "image/png"
1010
+ };
1011
+ warnings.push({
1012
+ type: "unsupported",
1013
+ feature: "URL-based image input",
1014
+ details: "Vertex AI video models require base64-encoded images or GCS URIs. URL will be ignored."
1015
+ });
1016
+ return;
1017
+ }
1018
+ return {
1019
+ bytesBase64Encoded: typeof file.data === "string" ? file.data : convertUint8ArrayToBase64(file.data),
1020
+ mimeType: file.mediaType || "image/png"
1021
+ };
1328
1022
  }
1329
1023
  function convertInputReferenceImage(file, warnings) {
1330
- const image = convertFileToVertexImage(file, warnings);
1331
- return image != null ? { image, referenceType: "asset" } : void 0;
1024
+ const image = convertFileToVertexImage(file, warnings);
1025
+ return image != null ? {
1026
+ image,
1027
+ referenceType: "asset"
1028
+ } : void 0;
1332
1029
  }
1333
1030
  var GoogleVertexVideoModel = class {
1334
- constructor(modelId, config) {
1335
- this.modelId = modelId;
1336
- this.config = config;
1337
- this.specificationVersion = "v4";
1338
- }
1339
- get provider() {
1340
- return this.config.provider;
1341
- }
1342
- get maxVideosPerCall() {
1343
- return 4;
1344
- }
1345
- async buildRequest(options) {
1346
- const warnings = [];
1347
- const googleVertexOptions = await parseProviderOptions4({
1348
- provider: "googleVertex",
1349
- providerOptions: options.providerOptions,
1350
- schema: googleVertexVideoModelOptionsSchema
1351
- }) ?? await parseProviderOptions4({
1352
- provider: "vertex",
1353
- providerOptions: options.providerOptions,
1354
- schema: googleVertexVideoModelOptionsSchema
1355
- });
1356
- const instances = [{}];
1357
- const instance = instances[0];
1358
- if (options.prompt != null) {
1359
- instance.prompt = options.prompt;
1360
- }
1361
- const startImage = resolveStartImage(options);
1362
- if (startImage != null) {
1363
- const image = convertFileToVertexImage(startImage, warnings);
1364
- if (image != null) {
1365
- instance.image = image;
1366
- }
1367
- }
1368
- const lastFrameImage = getLastFrameImage(options);
1369
- if (lastFrameImage != null) {
1370
- const lastFrame = convertFileToVertexImage(lastFrameImage, warnings);
1371
- if (lastFrame != null) {
1372
- instance.lastFrame = lastFrame;
1373
- }
1374
- }
1375
- const inputReferences = getInputReferences(options);
1376
- if (inputReferences != null) {
1377
- instance.referenceImages = inputReferences.flatMap((reference) => {
1378
- const converted = convertInputReferenceImage(reference, warnings);
1379
- return converted != null ? [converted] : [];
1380
- });
1381
- } else if (googleVertexOptions?.referenceImages != null) {
1382
- instance.referenceImages = googleVertexOptions.referenceImages;
1383
- }
1384
- const parameters = {
1385
- sampleCount: options.n
1386
- };
1387
- if (options.aspectRatio) {
1388
- parameters.aspectRatio = options.aspectRatio;
1389
- }
1390
- if (options.resolution) {
1391
- const resolutionMap = {
1392
- "1280x720": "720p",
1393
- "1920x1080": "1080p",
1394
- "3840x2160": "4k"
1395
- };
1396
- parameters.resolution = resolutionMap[options.resolution] || options.resolution;
1397
- }
1398
- if (options.duration) {
1399
- parameters.durationSeconds = options.duration;
1400
- }
1401
- if (options.seed) {
1402
- parameters.seed = options.seed;
1403
- }
1404
- const generateAudio = options.generateAudio ?? googleVertexOptions?.generateAudio;
1405
- if (generateAudio != null) {
1406
- parameters.generateAudio = generateAudio;
1407
- }
1408
- if (googleVertexOptions != null) {
1409
- const opts = googleVertexOptions;
1410
- if (opts.personGeneration !== void 0 && opts.personGeneration !== null) {
1411
- parameters.personGeneration = opts.personGeneration;
1412
- }
1413
- if (opts.negativePrompt !== void 0 && opts.negativePrompt !== null) {
1414
- parameters.negativePrompt = opts.negativePrompt;
1415
- }
1416
- if (opts.gcsOutputDirectory !== void 0 && opts.gcsOutputDirectory !== null) {
1417
- parameters.gcsOutputDirectory = opts.gcsOutputDirectory;
1418
- }
1419
- for (const [key, value] of Object.entries(opts)) {
1420
- if (![
1421
- "pollIntervalMs",
1422
- "pollTimeoutMs",
1423
- "personGeneration",
1424
- "negativePrompt",
1425
- "generateAudio",
1426
- "gcsOutputDirectory",
1427
- "referenceImages"
1428
- ].includes(key)) {
1429
- parameters[key] = value;
1430
- }
1431
- }
1432
- }
1433
- return { instances, parameters, warnings, googleVertexOptions };
1434
- }
1435
- buildCompletedResult({
1436
- finalOperation,
1437
- responseHeaders,
1438
- warnings,
1439
- currentDate
1440
- }) {
1441
- const response = finalOperation.response;
1442
- if (!response?.videos || response.videos.length === 0) {
1443
- throw new AISDKError({
1444
- name: "VERTEX_VIDEO_GENERATION_ERROR",
1445
- message: `No videos in response. Response: ${JSON.stringify(finalOperation)}`
1446
- });
1447
- }
1448
- const videos = [];
1449
- const videoMetadata = [];
1450
- for (const video of response.videos) {
1451
- if (video.bytesBase64Encoded) {
1452
- videos.push({
1453
- type: "base64",
1454
- data: video.bytesBase64Encoded,
1455
- mediaType: video.mimeType || "video/mp4"
1456
- });
1457
- videoMetadata.push({
1458
- mimeType: video.mimeType
1459
- });
1460
- } else if (video.gcsUri) {
1461
- videos.push({
1462
- type: "url",
1463
- url: video.gcsUri,
1464
- mediaType: video.mimeType || "video/mp4"
1465
- });
1466
- videoMetadata.push({
1467
- gcsUri: video.gcsUri,
1468
- mimeType: video.mimeType
1469
- });
1470
- }
1471
- }
1472
- if (videos.length === 0) {
1473
- throw new AISDKError({
1474
- name: "VERTEX_VIDEO_GENERATION_ERROR",
1475
- message: "No valid videos in response"
1476
- });
1477
- }
1478
- return {
1479
- status: "completed",
1480
- videos,
1481
- warnings,
1482
- response: {
1483
- timestamp: currentDate,
1484
- modelId: this.modelId,
1485
- headers: responseHeaders
1486
- },
1487
- providerMetadata: /* @__PURE__ */ (() => {
1488
- const payload = { videos: videoMetadata };
1489
- return {
1490
- googleVertex: payload,
1491
- // Legacy keys preserved for backward compatibility.
1492
- "google-vertex": payload,
1493
- vertex: payload
1494
- };
1495
- })()
1496
- };
1497
- }
1498
- async doStart(options) {
1499
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
1500
- const { instances, parameters, warnings } = await this.buildRequest(options);
1501
- const { value: operation, responseHeaders } = await postJsonToApi5({
1502
- url: `${this.config.baseURL}/models/${this.modelId}:predictLongRunning`,
1503
- headers: combineHeaders5(
1504
- await resolve5(this.config.headers),
1505
- options.headers
1506
- ),
1507
- body: {
1508
- instances,
1509
- parameters
1510
- },
1511
- successfulResponseHandler: createJsonResponseHandler5(
1512
- googleVertexOperationSchema
1513
- ),
1514
- failedResponseHandler: googleVertexFailedResponseHandler,
1515
- abortSignal: options.abortSignal,
1516
- fetch: this.config.fetch
1517
- });
1518
- const operationName = operation.name;
1519
- if (!operationName) {
1520
- throw new AISDKError({
1521
- name: "VERTEX_VIDEO_GENERATION_ERROR",
1522
- message: "No operation name returned from API"
1523
- });
1524
- }
1525
- return {
1526
- operation: { operationName },
1527
- warnings,
1528
- response: {
1529
- timestamp: currentDate,
1530
- modelId: this.modelId,
1531
- headers: responseHeaders
1532
- }
1533
- };
1534
- }
1535
- async doStatus(options) {
1536
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
1537
- const { operationName } = options.operation;
1538
- const { value: statusOperation, responseHeaders } = await postJsonToApi5({
1539
- url: `${this.config.baseURL}/models/${this.modelId}:fetchPredictOperation`,
1540
- headers: combineHeaders5(
1541
- await resolve5(this.config.headers),
1542
- options.headers
1543
- ),
1544
- body: {
1545
- operationName
1546
- },
1547
- successfulResponseHandler: createJsonResponseHandler5(
1548
- googleVertexOperationSchema
1549
- ),
1550
- failedResponseHandler: googleVertexFailedResponseHandler,
1551
- abortSignal: options.abortSignal,
1552
- fetch: this.config.fetch
1553
- });
1554
- if (!statusOperation.done) {
1555
- return {
1556
- status: "pending",
1557
- response: {
1558
- timestamp: currentDate,
1559
- modelId: this.modelId,
1560
- headers: responseHeaders
1561
- }
1562
- };
1563
- }
1564
- if (statusOperation.error) {
1565
- return {
1566
- status: "error",
1567
- error: `Video generation failed: ${statusOperation.error.message}`,
1568
- response: {
1569
- timestamp: currentDate,
1570
- modelId: this.modelId,
1571
- headers: responseHeaders
1572
- }
1573
- };
1574
- }
1575
- return this.buildCompletedResult({
1576
- finalOperation: statusOperation,
1577
- responseHeaders,
1578
- warnings: [],
1579
- currentDate
1580
- });
1581
- }
1031
+ get provider() {
1032
+ return this.config.provider;
1033
+ }
1034
+ get maxVideosPerCall() {
1035
+ return 4;
1036
+ }
1037
+ constructor(modelId, config) {
1038
+ this.modelId = modelId;
1039
+ this.config = config;
1040
+ this.specificationVersion = "v4";
1041
+ }
1042
+ async buildRequest(options) {
1043
+ const warnings = [];
1044
+ const googleVertexOptions = await parseProviderOptions({
1045
+ provider: "googleVertex",
1046
+ providerOptions: options.providerOptions,
1047
+ schema: googleVertexVideoModelOptionsSchema
1048
+ }) ?? await parseProviderOptions({
1049
+ provider: "vertex",
1050
+ providerOptions: options.providerOptions,
1051
+ schema: googleVertexVideoModelOptionsSchema
1052
+ });
1053
+ const instances = [{}];
1054
+ const instance = instances[0];
1055
+ if (options.prompt != null) instance.prompt = options.prompt;
1056
+ const startImage = resolveStartImage(options);
1057
+ if (startImage != null) {
1058
+ const image = convertFileToVertexImage(startImage, warnings);
1059
+ if (image != null) instance.image = image;
1060
+ }
1061
+ const lastFrameImage = getLastFrameImage(options);
1062
+ if (lastFrameImage != null) {
1063
+ const lastFrame = convertFileToVertexImage(lastFrameImage, warnings);
1064
+ if (lastFrame != null) instance.lastFrame = lastFrame;
1065
+ }
1066
+ const inputReferences = getInputReferences(options);
1067
+ if (inputReferences != null) instance.referenceImages = inputReferences.flatMap((reference) => {
1068
+ const converted = convertInputReferenceImage(reference, warnings);
1069
+ return converted != null ? [converted] : [];
1070
+ });
1071
+ else if (googleVertexOptions?.referenceImages != null) instance.referenceImages = googleVertexOptions.referenceImages;
1072
+ const parameters = { sampleCount: options.n };
1073
+ if (options.aspectRatio) parameters.aspectRatio = options.aspectRatio;
1074
+ if (options.resolution) parameters.resolution = {
1075
+ "1280x720": "720p",
1076
+ "1920x1080": "1080p",
1077
+ "3840x2160": "4k"
1078
+ }[options.resolution] || options.resolution;
1079
+ if (options.duration) parameters.durationSeconds = options.duration;
1080
+ if (options.seed) parameters.seed = options.seed;
1081
+ const generateAudio = options.generateAudio ?? googleVertexOptions?.generateAudio;
1082
+ if (generateAudio != null) parameters.generateAudio = generateAudio;
1083
+ if (googleVertexOptions != null) {
1084
+ const opts = googleVertexOptions;
1085
+ if (opts.personGeneration !== void 0 && opts.personGeneration !== null) parameters.personGeneration = opts.personGeneration;
1086
+ if (opts.negativePrompt !== void 0 && opts.negativePrompt !== null) parameters.negativePrompt = opts.negativePrompt;
1087
+ if (opts.gcsOutputDirectory !== void 0 && opts.gcsOutputDirectory !== null) parameters.gcsOutputDirectory = opts.gcsOutputDirectory;
1088
+ for (const [key, value] of Object.entries(opts)) if (![
1089
+ "pollIntervalMs",
1090
+ "pollTimeoutMs",
1091
+ "personGeneration",
1092
+ "negativePrompt",
1093
+ "generateAudio",
1094
+ "gcsOutputDirectory",
1095
+ "referenceImages"
1096
+ ].includes(key)) parameters[key] = value;
1097
+ }
1098
+ return {
1099
+ instances,
1100
+ parameters,
1101
+ warnings,
1102
+ googleVertexOptions
1103
+ };
1104
+ }
1105
+ buildCompletedResult({ finalOperation, responseHeaders, warnings, currentDate }) {
1106
+ const response = finalOperation.response;
1107
+ if (!response?.videos || response.videos.length === 0) throw new AISDKError({
1108
+ name: "VERTEX_VIDEO_GENERATION_ERROR",
1109
+ message: `No videos in response. Response: ${JSON.stringify(finalOperation)}`
1110
+ });
1111
+ const videos = [];
1112
+ const videoMetadata = [];
1113
+ for (const video of response.videos) if (video.bytesBase64Encoded) {
1114
+ videos.push({
1115
+ type: "base64",
1116
+ data: video.bytesBase64Encoded,
1117
+ mediaType: video.mimeType || "video/mp4"
1118
+ });
1119
+ videoMetadata.push({ mimeType: video.mimeType });
1120
+ } else if (video.gcsUri) {
1121
+ videos.push({
1122
+ type: "url",
1123
+ url: video.gcsUri,
1124
+ mediaType: video.mimeType || "video/mp4"
1125
+ });
1126
+ videoMetadata.push({
1127
+ gcsUri: video.gcsUri,
1128
+ mimeType: video.mimeType
1129
+ });
1130
+ }
1131
+ if (videos.length === 0) throw new AISDKError({
1132
+ name: "VERTEX_VIDEO_GENERATION_ERROR",
1133
+ message: "No valid videos in response"
1134
+ });
1135
+ return {
1136
+ status: "completed",
1137
+ videos,
1138
+ warnings,
1139
+ response: {
1140
+ timestamp: currentDate,
1141
+ modelId: this.modelId,
1142
+ headers: responseHeaders
1143
+ },
1144
+ providerMetadata: (() => {
1145
+ const payload = { videos: videoMetadata };
1146
+ return {
1147
+ googleVertex: payload,
1148
+ "google-vertex": payload,
1149
+ vertex: payload
1150
+ };
1151
+ })()
1152
+ };
1153
+ }
1154
+ async doStart(options) {
1155
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
1156
+ const { instances, parameters, warnings } = await this.buildRequest(options);
1157
+ const { value: operation, responseHeaders } = await postJsonToApi({
1158
+ url: `${this.config.baseURL}/models/${this.modelId}:predictLongRunning`,
1159
+ headers: combineHeaders(await resolve(this.config.headers), options.headers),
1160
+ body: {
1161
+ instances,
1162
+ parameters
1163
+ },
1164
+ successfulResponseHandler: createJsonResponseHandler(googleVertexOperationSchema),
1165
+ failedResponseHandler: googleVertexFailedResponseHandler,
1166
+ abortSignal: options.abortSignal,
1167
+ fetch: this.config.fetch
1168
+ });
1169
+ const operationName = operation.name;
1170
+ if (!operationName) throw new AISDKError({
1171
+ name: "VERTEX_VIDEO_GENERATION_ERROR",
1172
+ message: "No operation name returned from API"
1173
+ });
1174
+ return {
1175
+ operation: { operationName },
1176
+ warnings,
1177
+ response: {
1178
+ timestamp: currentDate,
1179
+ modelId: this.modelId,
1180
+ headers: responseHeaders
1181
+ }
1182
+ };
1183
+ }
1184
+ async doStatus(options) {
1185
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
1186
+ const { operationName } = options.operation;
1187
+ const { value: statusOperation, responseHeaders } = await postJsonToApi({
1188
+ url: `${this.config.baseURL}/models/${this.modelId}:fetchPredictOperation`,
1189
+ headers: combineHeaders(await resolve(this.config.headers), options.headers),
1190
+ body: { operationName },
1191
+ successfulResponseHandler: createJsonResponseHandler(googleVertexOperationSchema),
1192
+ failedResponseHandler: googleVertexFailedResponseHandler,
1193
+ abortSignal: options.abortSignal,
1194
+ fetch: this.config.fetch
1195
+ });
1196
+ if (!statusOperation.done) return {
1197
+ status: "pending",
1198
+ response: {
1199
+ timestamp: currentDate,
1200
+ modelId: this.modelId,
1201
+ headers: responseHeaders
1202
+ }
1203
+ };
1204
+ if (statusOperation.error) return {
1205
+ status: "error",
1206
+ error: `Video generation failed: ${statusOperation.error.message}`,
1207
+ response: {
1208
+ timestamp: currentDate,
1209
+ modelId: this.modelId,
1210
+ headers: responseHeaders
1211
+ }
1212
+ };
1213
+ return this.buildCompletedResult({
1214
+ finalOperation: statusOperation,
1215
+ responseHeaders,
1216
+ warnings: [],
1217
+ currentDate
1218
+ });
1219
+ }
1582
1220
  };
1583
- var googleVertexOperationSchema = z10.object({
1584
- name: z10.string().nullish(),
1585
- done: z10.boolean().nullish(),
1586
- error: z10.object({
1587
- code: z10.number().nullish(),
1588
- message: z10.string(),
1589
- status: z10.string().nullish()
1590
- }).nullish(),
1591
- response: z10.object({
1592
- videos: z10.array(
1593
- z10.object({
1594
- bytesBase64Encoded: z10.string().nullish(),
1595
- gcsUri: z10.string().nullish(),
1596
- mimeType: z10.string().nullish()
1597
- })
1598
- ).nullish(),
1599
- raiMediaFilteredCount: z10.number().nullish()
1600
- }).nullish()
1221
+ const googleVertexOperationSchema = z.object({
1222
+ name: z.string().nullish(),
1223
+ done: z.boolean().nullish(),
1224
+ error: z.object({
1225
+ code: z.number().nullish(),
1226
+ message: z.string(),
1227
+ status: z.string().nullish()
1228
+ }).nullish(),
1229
+ response: z.object({
1230
+ videos: z.array(z.object({
1231
+ bytesBase64Encoded: z.string().nullish(),
1232
+ gcsUri: z.string().nullish(),
1233
+ mimeType: z.string().nullish()
1234
+ })).nullish(),
1235
+ raiMediaFilteredCount: z.number().nullish()
1236
+ }).nullish()
1601
1237
  });
1602
-
1603
- // src/google-vertex-provider-base.ts
1604
- var EXPRESS_MODE_BASE_URL = "https://aiplatform.googleapis.com/v1/publishers/google";
1605
- var ENDPOINT_MODEL_PREFIX = "endpoints/";
1238
+ //#endregion
1239
+ //#region src/google-vertex-provider-base.ts
1240
+ const EXPRESS_MODE_BASE_URL = "https://aiplatform.googleapis.com/v1/publishers/google";
1241
+ const ENDPOINT_MODEL_PREFIX = "endpoints/";
1606
1242
  function isEndpointModelId(modelId) {
1607
- return modelId.startsWith(ENDPOINT_MODEL_PREFIX);
1243
+ return modelId.startsWith(ENDPOINT_MODEL_PREFIX);
1608
1244
  }
1609
1245
  function createExpressModeFetch(apiKey, customFetch) {
1610
- return async (url, init) => {
1611
- const modifiedInit = {
1612
- ...init,
1613
- headers: {
1614
- ...init?.headers ? normalizeHeaders(init.headers) : {},
1615
- "x-goog-api-key": apiKey
1616
- }
1617
- };
1618
- return (customFetch ?? fetch)(url.toString(), modifiedInit);
1619
- };
1246
+ return async (url, init) => {
1247
+ const modifiedInit = {
1248
+ ...init,
1249
+ headers: {
1250
+ ...init?.headers ? normalizeHeaders(init.headers) : {},
1251
+ "x-goog-api-key": apiKey
1252
+ }
1253
+ };
1254
+ return (customFetch ?? fetch)(url.toString(), modifiedInit);
1255
+ };
1620
1256
  }
1257
+ /**
1258
+ * Create a Google Vertex AI provider instance.
1259
+ */
1260
+ function createGoogleVertex$1(options = {}) {
1261
+ const apiKey = loadOptionalSetting({
1262
+ settingValue: options.apiKey,
1263
+ environmentVariableName: "GOOGLE_VERTEX_API_KEY"
1264
+ });
1265
+ const loadGoogleVertexProject = () => loadSetting({
1266
+ settingValue: options.project,
1267
+ settingName: "project",
1268
+ environmentVariableName: "GOOGLE_VERTEX_PROJECT",
1269
+ description: "Google Vertex project"
1270
+ });
1271
+ const loadGoogleVertexLocation = () => loadSetting({
1272
+ settingValue: options.location,
1273
+ settingName: "location",
1274
+ environmentVariableName: "GOOGLE_VERTEX_LOCATION",
1275
+ description: "Google Vertex location"
1276
+ });
1277
+ const loadBaseURL = ({ endpoint = false } = {}) => {
1278
+ if (apiKey) return withoutTrailingSlash(options.baseURL) ?? EXPRESS_MODE_BASE_URL;
1279
+ const region = loadGoogleVertexLocation();
1280
+ const project = loadGoogleVertexProject();
1281
+ const getHost = () => {
1282
+ if (region === "global") return "aiplatform.googleapis.com";
1283
+ else if (region === "eu" || region === "us") return `aiplatform.${region}.rep.googleapis.com`;
1284
+ else return `${region}-aiplatform.googleapis.com`;
1285
+ };
1286
+ return withoutTrailingSlash(options.baseURL) ?? `https://${getHost()}/v1beta1/projects/${project}/locations/${region}${endpoint ? "" : "/publishers/google"}`;
1287
+ };
1288
+ const createConfig = (name, { endpoint = false } = {}) => {
1289
+ const getHeaders = async () => {
1290
+ const originalHeaders = await resolve(options.headers ?? {});
1291
+ return withUserAgentSuffix(originalHeaders, `ai-sdk-google-vertex/${VERSION}`);
1292
+ };
1293
+ return {
1294
+ provider: `google.vertex.${name}`,
1295
+ headers: getHeaders,
1296
+ fetch: apiKey ? createExpressModeFetch(apiKey, options.fetch) : options.fetch,
1297
+ baseURL: loadBaseURL({ endpoint })
1298
+ };
1299
+ };
1300
+ const createChatModel = (modelId) => {
1301
+ const endpoint = isEndpointModelId(modelId);
1302
+ if (endpoint && apiKey) throw new Error("Google Vertex tuned models do not support Express Mode API keys. Use standard Google Cloud credentials instead.");
1303
+ return new GoogleLanguageModel(modelId, {
1304
+ ...createConfig("chat", { endpoint }),
1305
+ generateId: options.generateId ?? generateId,
1306
+ supportedUrls: () => ({ "*": [/^https?:\/\/.*$/, /^gs:\/\/.*$/] }),
1307
+ downloadToolResultFiles: {
1308
+ maxBytes: options.toolResultDownloads?.maxBytes ?? 7340032,
1309
+ supportsGoogleCloudStorageUrls: true
1310
+ }
1311
+ });
1312
+ };
1313
+ const createInteractionsModel = (modelIdOrAgent) => {
1314
+ if (apiKey) throw new Error("Google Vertex Interactions models do not support Express Mode API keys. Use standard Google Cloud credentials instead.");
1315
+ return new GoogleInteractionsLanguageModel(modelIdOrAgent, {
1316
+ ...createConfig("interactions", { endpoint: true }),
1317
+ generateId: options.generateId ?? generateId
1318
+ });
1319
+ };
1320
+ const createEmbeddingModel = (modelId) => new GoogleVertexEmbeddingModel(modelId, createConfig("embedding"));
1321
+ const createImageModel = (modelId) => new GoogleVertexImageModel(modelId, {
1322
+ ...createConfig("image"),
1323
+ generateId: options.generateId ?? generateId
1324
+ });
1325
+ const createVideoModel = (modelId) => new GoogleVertexVideoModel(modelId, {
1326
+ ...createConfig("video"),
1327
+ generateId: options.generateId ?? generateId
1328
+ });
1329
+ const createSpeechModel = (modelId) => {
1330
+ if (modelId.startsWith("chirp")) {
1331
+ if (apiKey) throw new Error("Google Vertex Chirp speech models do not support Express Mode API keys. Use standard Google Cloud credentials instead.");
1332
+ const config = createConfig("speech");
1333
+ return new GoogleVertexCloudTTSSpeechModel(modelId, {
1334
+ provider: config.provider,
1335
+ headers: config.headers,
1336
+ fetch: config.fetch
1337
+ });
1338
+ }
1339
+ return new GoogleSpeechModel(modelId, createConfig("speech"));
1340
+ };
1341
+ const createTranscriptionModel = (modelId) => {
1342
+ if (apiKey) throw new Error("Google Vertex transcription models do not support Express Mode API keys. Use standard Google Cloud credentials instead.");
1343
+ const config = createConfig("transcription");
1344
+ if (modelId.startsWith("gemini")) return new GoogleVertexGeminiTranscriptionModel(modelId, {
1345
+ provider: config.provider,
1346
+ baseURL: loadBaseURL(),
1347
+ headers: config.headers,
1348
+ fetch: config.fetch,
1349
+ webSocket: options.webSocket,
1350
+ project: loadGoogleVertexProject(),
1351
+ location: loadGoogleVertexLocation()
1352
+ });
1353
+ return new GoogleVertexTranscriptionModel(modelId, {
1354
+ provider: config.provider,
1355
+ headers: config.headers,
1356
+ fetch: config.fetch,
1357
+ project: loadGoogleVertexProject(),
1358
+ location: loadGoogleVertexLocation()
1359
+ });
1360
+ };
1361
+ const provider = function(modelId) {
1362
+ if (new.target) throw new Error("The Google Vertex AI model function cannot be called with the new keyword.");
1363
+ return createChatModel(modelId);
1364
+ };
1365
+ provider.specificationVersion = "v4";
1366
+ provider.languageModel = createChatModel;
1367
+ provider.interactions = createInteractionsModel;
1368
+ provider.embeddingModel = createEmbeddingModel;
1369
+ provider.textEmbeddingModel = createEmbeddingModel;
1370
+ provider.image = createImageModel;
1371
+ provider.imageModel = createImageModel;
1372
+ provider.video = createVideoModel;
1373
+ provider.videoModel = createVideoModel;
1374
+ provider.speech = createSpeechModel;
1375
+ provider.speechModel = createSpeechModel;
1376
+ provider.transcription = createTranscriptionModel;
1377
+ provider.transcriptionModel = createTranscriptionModel;
1378
+ provider.tools = googleVertexTools;
1379
+ return provider;
1380
+ }
1381
+ //#endregion
1382
+ //#region src/google-vertex-provider.ts
1621
1383
  function createGoogleVertex(options = {}) {
1622
- const apiKey = loadOptionalSetting({
1623
- settingValue: options.apiKey,
1624
- environmentVariableName: "GOOGLE_VERTEX_API_KEY"
1625
- });
1626
- const loadGoogleVertexProject = () => loadSetting({
1627
- settingValue: options.project,
1628
- settingName: "project",
1629
- environmentVariableName: "GOOGLE_VERTEX_PROJECT",
1630
- description: "Google Vertex project"
1631
- });
1632
- const loadGoogleVertexLocation = () => loadSetting({
1633
- settingValue: options.location,
1634
- settingName: "location",
1635
- environmentVariableName: "GOOGLE_VERTEX_LOCATION",
1636
- description: "Google Vertex location"
1637
- });
1638
- const loadBaseURL = ({ endpoint = false } = {}) => {
1639
- if (apiKey) {
1640
- return withoutTrailingSlash(options.baseURL) ?? EXPRESS_MODE_BASE_URL;
1641
- }
1642
- const region = loadGoogleVertexLocation();
1643
- const project = loadGoogleVertexProject();
1644
- const getHost = () => {
1645
- if (region === "global") {
1646
- return "aiplatform.googleapis.com";
1647
- } else if (region === "eu" || region === "us") {
1648
- return `aiplatform.${region}.rep.googleapis.com`;
1649
- } else {
1650
- return `${region}-aiplatform.googleapis.com`;
1651
- }
1652
- };
1653
- return withoutTrailingSlash(options.baseURL) ?? `https://${getHost()}/v1beta1/projects/${project}/locations/${region}${endpoint ? "" : "/publishers/google"}`;
1654
- };
1655
- const createConfig = (name, { endpoint = false } = {}) => {
1656
- const getHeaders = async () => {
1657
- const originalHeaders = await resolve6(options.headers ?? {});
1658
- return withUserAgentSuffix(
1659
- originalHeaders,
1660
- `ai-sdk/google-vertex/${VERSION}`
1661
- );
1662
- };
1663
- return {
1664
- provider: `google.vertex.${name}`,
1665
- headers: getHeaders,
1666
- fetch: apiKey ? createExpressModeFetch(apiKey, options.fetch) : options.fetch,
1667
- baseURL: loadBaseURL({ endpoint })
1668
- };
1669
- };
1670
- const createChatModel = (modelId) => {
1671
- const endpoint = isEndpointModelId(modelId);
1672
- if (endpoint && apiKey) {
1673
- throw new Error(
1674
- "Google Vertex tuned models do not support Express Mode API keys. Use standard Google Cloud credentials instead."
1675
- );
1676
- }
1677
- return new GoogleLanguageModel2(modelId, {
1678
- ...createConfig("chat", { endpoint }),
1679
- generateId: options.generateId ?? generateId,
1680
- supportedUrls: () => ({
1681
- "*": [
1682
- // HTTP URLs:
1683
- /^https?:\/\/.*$/,
1684
- // Google Cloud Storage URLs:
1685
- /^gs:\/\/.*$/
1686
- ]
1687
- }),
1688
- downloadToolResultFiles: {
1689
- maxBytes: options.toolResultDownloads?.maxBytes ?? 7 * 1024 * 1024,
1690
- supportsGoogleCloudStorageUrls: true
1691
- }
1692
- });
1693
- };
1694
- const createInteractionsModel = (modelIdOrAgent) => {
1695
- if (apiKey) {
1696
- throw new Error(
1697
- "Google Vertex Interactions models do not support Express Mode API keys. Use standard Google Cloud credentials instead."
1698
- );
1699
- }
1700
- return new GoogleInteractionsLanguageModel(
1701
- modelIdOrAgent,
1702
- {
1703
- // The Interactions API is a location-scoped resource
1704
- // (`.../locations/{region}/interactions`), so it uses the
1705
- // endpoint-style base URL without the `/publishers/google` suffix that
1706
- // the base-model paths carry.
1707
- ...createConfig("interactions", { endpoint: true }),
1708
- generateId: options.generateId ?? generateId
1709
- }
1710
- );
1711
- };
1712
- const createEmbeddingModel = (modelId) => new GoogleVertexEmbeddingModel(modelId, createConfig("embedding"));
1713
- const createImageModel = (modelId) => new GoogleVertexImageModel(modelId, {
1714
- ...createConfig("image"),
1715
- generateId: options.generateId ?? generateId
1716
- });
1717
- const createVideoModel = (modelId) => new GoogleVertexVideoModel(modelId, {
1718
- ...createConfig("video"),
1719
- generateId: options.generateId ?? generateId
1720
- });
1721
- const createSpeechModel = (modelId) => {
1722
- if (modelId.startsWith("chirp")) {
1723
- if (apiKey) {
1724
- throw new Error(
1725
- "Google Vertex Chirp speech models do not support Express Mode API keys. Use standard Google Cloud credentials instead."
1726
- );
1727
- }
1728
- const config = createConfig("speech");
1729
- return new GoogleVertexCloudTTSSpeechModel(modelId, {
1730
- provider: config.provider,
1731
- headers: config.headers,
1732
- fetch: config.fetch
1733
- });
1734
- }
1735
- return new GoogleSpeechModel(modelId, createConfig("speech"));
1736
- };
1737
- const createTranscriptionModel = (modelId) => {
1738
- if (apiKey) {
1739
- throw new Error(
1740
- "Google Vertex transcription models do not support Express Mode API keys. Use standard Google Cloud credentials instead."
1741
- );
1742
- }
1743
- const config = createConfig("transcription");
1744
- if (modelId.startsWith("gemini")) {
1745
- return new GoogleVertexGeminiTranscriptionModel(modelId, {
1746
- provider: config.provider,
1747
- baseURL: loadBaseURL(),
1748
- headers: config.headers,
1749
- fetch: config.fetch,
1750
- webSocket: options.webSocket,
1751
- project: loadGoogleVertexProject(),
1752
- location: loadGoogleVertexLocation()
1753
- });
1754
- }
1755
- return new GoogleVertexTranscriptionModel(modelId, {
1756
- provider: config.provider,
1757
- headers: config.headers,
1758
- fetch: config.fetch,
1759
- project: loadGoogleVertexProject(),
1760
- location: loadGoogleVertexLocation()
1761
- });
1762
- };
1763
- const provider = function(modelId) {
1764
- if (new.target) {
1765
- throw new Error(
1766
- "The Google Vertex AI model function cannot be called with the new keyword."
1767
- );
1768
- }
1769
- return createChatModel(modelId);
1770
- };
1771
- provider.specificationVersion = "v4";
1772
- provider.languageModel = createChatModel;
1773
- provider.interactions = createInteractionsModel;
1774
- provider.embeddingModel = createEmbeddingModel;
1775
- provider.textEmbeddingModel = createEmbeddingModel;
1776
- provider.image = createImageModel;
1777
- provider.imageModel = createImageModel;
1778
- provider.video = createVideoModel;
1779
- provider.videoModel = createVideoModel;
1780
- provider.speech = createSpeechModel;
1781
- provider.speechModel = createSpeechModel;
1782
- provider.transcription = createTranscriptionModel;
1783
- provider.transcriptionModel = createTranscriptionModel;
1784
- provider.tools = googleVertexTools;
1785
- return provider;
1384
+ if (loadOptionalSetting({
1385
+ settingValue: options.apiKey,
1386
+ environmentVariableName: "GOOGLE_VERTEX_API_KEY"
1387
+ })) return createGoogleVertex$1(options);
1388
+ const generateAuthToken = createAuthTokenGenerator(options.project == null ? options.googleAuthOptions : {
1389
+ projectId: options.project,
1390
+ ...options.googleAuthOptions
1391
+ });
1392
+ return createGoogleVertex$1({
1393
+ ...options,
1394
+ headers: async () => ({
1395
+ Authorization: `Bearer ${await generateAuthToken()}`,
1396
+ ...await resolve(options.headers)
1397
+ })
1398
+ });
1786
1399
  }
1400
+ /**
1401
+ * Default Google Vertex AI provider instance.
1402
+ */
1403
+ const googleVertex = createGoogleVertex();
1404
+ //#endregion
1405
+ export { GoogleVertexGeminiTranscriptionModel, VERSION, createGoogleVertex, createGoogleVertex as createVertex, googleVertex, googleVertex as vertex };
1787
1406
 
1788
- // src/google-vertex-provider.ts
1789
- function createGoogleVertex2(options = {}) {
1790
- const apiKey = loadOptionalSetting2({
1791
- settingValue: options.apiKey,
1792
- environmentVariableName: "GOOGLE_VERTEX_API_KEY"
1793
- });
1794
- if (apiKey) {
1795
- return createGoogleVertex(options);
1796
- }
1797
- const googleAuthOptions = options.project == null ? options.googleAuthOptions : {
1798
- projectId: options.project,
1799
- ...options.googleAuthOptions
1800
- };
1801
- const generateAuthToken = createAuthTokenGenerator(googleAuthOptions);
1802
- return createGoogleVertex({
1803
- ...options,
1804
- headers: async () => ({
1805
- Authorization: `Bearer ${await generateAuthToken()}`,
1806
- ...await resolve7(options.headers)
1807
- })
1808
- });
1809
- }
1810
- var googleVertex = createGoogleVertex2();
1811
- export {
1812
- GoogleVertexGeminiTranscriptionModel,
1813
- VERSION,
1814
- createGoogleVertex2 as createGoogleVertex,
1815
- createGoogleVertex2 as createVertex,
1816
- googleVertex,
1817
- googleVertex as vertex
1818
- };
1819
1407
  //# sourceMappingURL=index.js.map