@ai-sdk/google-vertex 5.0.98 → 5.0.100

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/CHANGELOG.md +28 -0
  2. package/dist/anthropic/edge/index.d.ts +84 -80
  3. package/dist/anthropic/edge/index.d.ts.map +1 -0
  4. package/dist/anthropic/edge/index.js +238 -272
  5. package/dist/anthropic/edge/index.js.map +1 -1
  6. package/dist/anthropic/index.d.ts +69 -66
  7. package/dist/anthropic/index.d.ts.map +1 -0
  8. package/dist/anthropic/index.js +172 -177
  9. package/dist/anthropic/index.js.map +1 -1
  10. package/dist/edge/index.d.ts +206 -194
  11. package/dist/edge/index.d.ts.map +1 -0
  12. package/dist/edge/index.js +1428 -1846
  13. package/dist/edge/index.js.map +1 -1
  14. package/dist/index.d.ts +265 -252
  15. package/dist/index.d.ts.map +1 -0
  16. package/dist/index.js +1369 -1761
  17. package/dist/index.js.map +1 -1
  18. package/dist/maas/edge/index.d.ts +62 -59
  19. package/dist/maas/edge/index.d.ts.map +1 -0
  20. package/dist/maas/edge/index.js +174 -196
  21. package/dist/maas/edge/index.js.map +1 -1
  22. package/dist/maas/index.d.ts +47 -45
  23. package/dist/maas/index.d.ts.map +1 -0
  24. package/dist/maas/index.js +109 -102
  25. package/dist/maas/index.js.map +1 -1
  26. package/dist/xai/edge/index.d.ts +77 -73
  27. package/dist/xai/edge/index.d.ts.map +1 -0
  28. package/dist/xai/edge/index.js +203 -235
  29. package/dist/xai/edge/index.js.map +1 -1
  30. package/dist/xai/index.d.ts +62 -59
  31. package/dist/xai/index.d.ts.map +1 -0
  32. package/dist/xai/index.js +138 -141
  33. package/dist/xai/index.js.map +1 -1
  34. package/docs/16-google-vertex.mdx +3 -3
  35. package/package.json +11 -11
  36. package/src/anthropic/google-vertex-anthropic-provider.ts +14 -4
  37. package/src/gemini-transcription/google-vertex-gemini-transcription-model.ts +5 -1
  38. package/src/google-vertex-provider-base.ts +28 -15
  39. package/src/google-vertex-transcription-model-options.ts +2 -1
  40. package/src/google-vertex-transcription-model.ts +13 -1
  41. package/src/maas/google-vertex-maas-provider.ts +18 -6
@@ -1,1907 +1,1489 @@
1
- // src/edge/google-vertex-provider-edge.ts
2
- import { loadOptionalSetting as loadOptionalSetting3, resolve as resolve7 } from "@ai-sdk/provider-utils";
3
-
4
- // src/google-vertex-provider-base.ts
5
- import {
6
- GoogleInteractionsLanguageModel,
7
- GoogleLanguageModel as GoogleLanguageModel2,
8
- GoogleSpeechModel
9
- } from "@ai-sdk/google/internal";
10
- import {
11
- generateId,
12
- loadOptionalSetting,
13
- loadSetting,
14
- normalizeHeaders,
15
- resolve as resolve6,
16
- withoutTrailingSlash,
17
- withUserAgentSuffix
18
- } from "@ai-sdk/provider-utils";
19
-
20
- // src/version.ts
21
- var VERSION = true ? "5.0.98" : "0.0.0-test";
22
-
23
- // src/google-vertex-embedding-model.ts
24
- import {
25
- TooManyEmbeddingValuesForCallError
26
- } from "@ai-sdk/provider";
27
- import {
28
- combineHeaders,
29
- createJsonResponseHandler,
30
- postJsonToApi,
31
- resolve,
32
- parseProviderOptions,
33
- serializeModelOptions,
34
- WORKFLOW_SERIALIZE,
35
- WORKFLOW_DESERIALIZE
36
- } from "@ai-sdk/provider-utils";
37
- import { z as z3 } from "zod/v4";
38
-
39
- // src/google-vertex-error.ts
40
- import { createJsonErrorResponseHandler } from "@ai-sdk/provider-utils";
1
+ import { WORKFLOW_DESERIALIZE, WORKFLOW_SERIALIZE, combineHeaders, connectToWebSocket, convertBase64ToUint8Array, convertToBase64, convertUint8ArrayToBase64, createJsonErrorResponseHandler, createJsonResponseHandler, generateId, getRuntimeEnvironmentUserAgent, isValidHostnamePart, lazySchema, loadOptionalSetting, loadSetting, normalizeHeaders, parseProviderOptions, postJsonToApi, resolve, safeParseJSON, serializeModelOptions, waitForWebSocketBufferDrain, withUserAgentSuffix, withoutTrailingSlash, zodSchema } from "@ai-sdk/provider-utils";
2
+ import { GoogleInteractionsLanguageModel, GoogleLanguageModel, GoogleSpeechModel, googleTools } from "@ai-sdk/google/internal";
3
+ import { AISDKError, InvalidArgumentError, TooManyEmbeddingValuesForCallError } from "@ai-sdk/provider";
41
4
  import { z } from "zod/v4";
42
- var googleVertexErrorDataSchema = z.object({
43
- error: z.object({
44
- code: z.number().nullable(),
45
- message: z.string(),
46
- status: z.string()
47
- })
5
+ //#region src/version.ts
6
+ const VERSION = "5.0.100";
7
+ //#endregion
8
+ //#region src/google-vertex-error.ts
9
+ const googleVertexErrorDataSchema = z.object({ error: z.object({
10
+ code: z.number().nullable(),
11
+ message: z.string(),
12
+ status: z.string()
13
+ }) });
14
+ const googleVertexFailedResponseHandler = createJsonErrorResponseHandler({
15
+ errorSchema: googleVertexErrorDataSchema,
16
+ errorToMessage: (data) => data.error.message
48
17
  });
49
- var googleVertexFailedResponseHandler = createJsonErrorResponseHandler(
50
- {
51
- errorSchema: googleVertexErrorDataSchema,
52
- errorToMessage: (data) => data.error.message
53
- }
54
- );
55
-
56
- // src/google-vertex-embedding-model-options.ts
57
- import { z as z2 } from "zod/v4";
58
- var googleVertexEmbeddingModelOptions = z2.object({
59
- /**
60
- * Optional. Optional reduced dimension for the output embedding.
61
- * If set, excessive values in the output embedding are truncated from the end.
62
- */
63
- outputDimensionality: z2.number().optional(),
64
- /**
65
- * Optional. Specifies the task type for generating embeddings.
66
- * Supported task types:
67
- * - SEMANTIC_SIMILARITY: Optimized for text similarity.
68
- * - CLASSIFICATION: Optimized for text classification.
69
- * - CLUSTERING: Optimized for clustering texts based on similarity.
70
- * - RETRIEVAL_DOCUMENT: Optimized for document retrieval.
71
- * - RETRIEVAL_QUERY: Optimized for query-based retrieval.
72
- * - QUESTION_ANSWERING: Optimized for answering questions.
73
- * - FACT_VERIFICATION: Optimized for verifying factual information.
74
- * - CODE_RETRIEVAL_QUERY: Optimized for retrieving code blocks based on natural language queries.
75
- */
76
- taskType: z2.enum([
77
- "SEMANTIC_SIMILARITY",
78
- "CLASSIFICATION",
79
- "CLUSTERING",
80
- "RETRIEVAL_DOCUMENT",
81
- "RETRIEVAL_QUERY",
82
- "QUESTION_ANSWERING",
83
- "FACT_VERIFICATION",
84
- "CODE_RETRIEVAL_QUERY"
85
- ]).optional(),
86
- /**
87
- * Optional. The title of the document being embedded.
88
- * Only valid when task_type is set to 'RETRIEVAL_DOCUMENT'.
89
- * Helps the model produce better embeddings by providing additional context.
90
- */
91
- title: z2.string().optional(),
92
- /**
93
- * Optional. When set to true, input text will be truncated. When set to false,
94
- * an error is returned if the input text is longer than the maximum length supported by the model. Defaults to true.
95
- */
96
- autoTruncate: z2.boolean().optional()
18
+ //#endregion
19
+ //#region src/google-vertex-embedding-model-options.ts
20
+ const googleVertexEmbeddingModelOptions = z.object({
21
+ /**
22
+ * Optional. Optional reduced dimension for the output embedding.
23
+ * If set, excessive values in the output embedding are truncated from the end.
24
+ */
25
+ outputDimensionality: z.number().optional(),
26
+ /**
27
+ * Optional. Specifies the task type for generating embeddings.
28
+ * Supported task types:
29
+ * - SEMANTIC_SIMILARITY: Optimized for text similarity.
30
+ * - CLASSIFICATION: Optimized for text classification.
31
+ * - CLUSTERING: Optimized for clustering texts based on similarity.
32
+ * - RETRIEVAL_DOCUMENT: Optimized for document retrieval.
33
+ * - RETRIEVAL_QUERY: Optimized for query-based retrieval.
34
+ * - QUESTION_ANSWERING: Optimized for answering questions.
35
+ * - FACT_VERIFICATION: Optimized for verifying factual information.
36
+ * - CODE_RETRIEVAL_QUERY: Optimized for retrieving code blocks based on natural language queries.
37
+ */
38
+ taskType: z.enum([
39
+ "SEMANTIC_SIMILARITY",
40
+ "CLASSIFICATION",
41
+ "CLUSTERING",
42
+ "RETRIEVAL_DOCUMENT",
43
+ "RETRIEVAL_QUERY",
44
+ "QUESTION_ANSWERING",
45
+ "FACT_VERIFICATION",
46
+ "CODE_RETRIEVAL_QUERY"
47
+ ]).optional(),
48
+ /**
49
+ * Optional. The title of the document being embedded.
50
+ * Only valid when task_type is set to 'RETRIEVAL_DOCUMENT'.
51
+ * Helps the model produce better embeddings by providing additional context.
52
+ */
53
+ title: z.string().optional(),
54
+ /**
55
+ * Optional. When set to true, input text will be truncated. When set to false,
56
+ * an error is returned if the input text is longer than the maximum length supported by the model. Defaults to true.
57
+ */
58
+ autoTruncate: z.boolean().optional()
97
59
  });
98
-
99
- // src/google-vertex-embedding-model.ts
100
- var GoogleVertexEmbeddingModel = class _GoogleVertexEmbeddingModel {
101
- constructor(modelId, config) {
102
- this.specificationVersion = "v4";
103
- this.supportsParallelCalls = true;
104
- this.modelId = modelId;
105
- this.config = config;
106
- }
107
- static [WORKFLOW_SERIALIZE](model) {
108
- return serializeModelOptions({
109
- modelId: model.modelId,
110
- config: model.config
111
- });
112
- }
113
- static [WORKFLOW_DESERIALIZE](options) {
114
- return new _GoogleVertexEmbeddingModel(options.modelId, options.config);
115
- }
116
- get provider() {
117
- return this.config.provider;
118
- }
119
- // gemini-embedding-2 models only support :embedContent (one value per call),
120
- // not the :predict batch endpoint. https://github.com/vercel/ai/issues/15853
121
- get maxEmbeddingsPerCall() {
122
- return usesEmbedContentEndpoint(this.modelId) ? 1 : 250;
123
- }
124
- async doEmbed({
125
- values,
126
- headers,
127
- abortSignal,
128
- providerOptions
129
- }) {
130
- let googleOptions = await parseProviderOptions({
131
- provider: "googleVertex",
132
- providerOptions,
133
- schema: googleVertexEmbeddingModelOptions
134
- });
135
- if (googleOptions == null) {
136
- googleOptions = await parseProviderOptions({
137
- provider: "vertex",
138
- providerOptions,
139
- schema: googleVertexEmbeddingModelOptions
140
- });
141
- }
142
- if (googleOptions == null) {
143
- googleOptions = await parseProviderOptions({
144
- provider: "google",
145
- providerOptions,
146
- schema: googleVertexEmbeddingModelOptions
147
- });
148
- }
149
- googleOptions = googleOptions ?? {};
150
- if (values.length > this.maxEmbeddingsPerCall) {
151
- throw new TooManyEmbeddingValuesForCallError({
152
- provider: this.provider,
153
- modelId: this.modelId,
154
- maxEmbeddingsPerCall: this.maxEmbeddingsPerCall,
155
- values
156
- });
157
- }
158
- const mergedHeaders = combineHeaders(
159
- this.config.headers ? await resolve(this.config.headers) : void 0,
160
- headers
161
- );
162
- if (usesEmbedContentEndpoint(this.modelId)) {
163
- const {
164
- responseHeaders: responseHeaders2,
165
- value: response2,
166
- rawValue: rawValue2
167
- } = await postJsonToApi({
168
- url: `${this.config.baseURL}/models/${this.modelId}:embedContent`,
169
- headers: mergedHeaders,
170
- body: {
171
- content: { parts: [{ text: values[0] }] },
172
- embedContentConfig: {
173
- outputDimensionality: googleOptions.outputDimensionality,
174
- taskType: googleOptions.taskType,
175
- title: googleOptions.title,
176
- autoTruncate: googleOptions.autoTruncate
177
- }
178
- },
179
- failedResponseHandler: googleVertexFailedResponseHandler,
180
- successfulResponseHandler: createJsonResponseHandler(
181
- googleVertexEmbedContentResponseSchema
182
- ),
183
- abortSignal,
184
- fetch: this.config.fetch
185
- });
186
- return {
187
- warnings: [],
188
- embeddings: [response2.embedding.values],
189
- usage: response2.usageMetadata?.promptTokenCount == null ? void 0 : { tokens: response2.usageMetadata.promptTokenCount },
190
- response: { headers: responseHeaders2, body: rawValue2 }
191
- };
192
- }
193
- const url = `${this.config.baseURL}/models/${this.modelId}:predict`;
194
- const {
195
- responseHeaders,
196
- value: response,
197
- rawValue
198
- } = await postJsonToApi({
199
- url,
200
- headers: mergedHeaders,
201
- body: {
202
- instances: values.map((value) => ({
203
- content: value,
204
- task_type: googleOptions.taskType,
205
- title: googleOptions.title
206
- })),
207
- parameters: {
208
- outputDimensionality: googleOptions.outputDimensionality,
209
- autoTruncate: googleOptions.autoTruncate
210
- }
211
- },
212
- failedResponseHandler: googleVertexFailedResponseHandler,
213
- successfulResponseHandler: createJsonResponseHandler(
214
- googleVertexTextEmbeddingResponseSchema
215
- ),
216
- abortSignal,
217
- fetch: this.config.fetch
218
- });
219
- return {
220
- warnings: [],
221
- embeddings: response.predictions.map(
222
- (prediction) => prediction.embeddings.values
223
- ),
224
- usage: {
225
- tokens: response.predictions.reduce(
226
- (tokenCount, prediction) => tokenCount + prediction.embeddings.statistics.token_count,
227
- 0
228
- )
229
- },
230
- response: { headers: responseHeaders, body: rawValue }
231
- };
232
- }
60
+ //#endregion
61
+ //#region src/google-vertex-embedding-model.ts
62
+ var GoogleVertexEmbeddingModel = class GoogleVertexEmbeddingModel {
63
+ static [WORKFLOW_SERIALIZE](model) {
64
+ return serializeModelOptions({
65
+ modelId: model.modelId,
66
+ config: model.config
67
+ });
68
+ }
69
+ static [WORKFLOW_DESERIALIZE](options) {
70
+ return new GoogleVertexEmbeddingModel(options.modelId, options.config);
71
+ }
72
+ get provider() {
73
+ return this.config.provider;
74
+ }
75
+ get maxEmbeddingsPerCall() {
76
+ return usesEmbedContentEndpoint(this.modelId) ? 1 : 250;
77
+ }
78
+ constructor(modelId, config) {
79
+ this.specificationVersion = "v4";
80
+ this.supportsParallelCalls = true;
81
+ this.modelId = modelId;
82
+ this.config = config;
83
+ }
84
+ async doEmbed({ values, headers, abortSignal, providerOptions }) {
85
+ let googleOptions = await parseProviderOptions({
86
+ provider: "googleVertex",
87
+ providerOptions,
88
+ schema: googleVertexEmbeddingModelOptions
89
+ });
90
+ if (googleOptions == null) googleOptions = await parseProviderOptions({
91
+ provider: "vertex",
92
+ providerOptions,
93
+ schema: googleVertexEmbeddingModelOptions
94
+ });
95
+ if (googleOptions == null) googleOptions = await parseProviderOptions({
96
+ provider: "google",
97
+ providerOptions,
98
+ schema: googleVertexEmbeddingModelOptions
99
+ });
100
+ googleOptions = googleOptions ?? {};
101
+ if (values.length > this.maxEmbeddingsPerCall) throw new TooManyEmbeddingValuesForCallError({
102
+ provider: this.provider,
103
+ modelId: this.modelId,
104
+ maxEmbeddingsPerCall: this.maxEmbeddingsPerCall,
105
+ values
106
+ });
107
+ const mergedHeaders = combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, headers);
108
+ if (usesEmbedContentEndpoint(this.modelId)) {
109
+ const { responseHeaders, value: response, rawValue } = await postJsonToApi({
110
+ url: `${this.config.baseURL}/models/${this.modelId}:embedContent`,
111
+ headers: mergedHeaders,
112
+ body: {
113
+ content: { parts: [{ text: values[0] }] },
114
+ embedContentConfig: {
115
+ outputDimensionality: googleOptions.outputDimensionality,
116
+ taskType: googleOptions.taskType,
117
+ title: googleOptions.title,
118
+ autoTruncate: googleOptions.autoTruncate
119
+ }
120
+ },
121
+ failedResponseHandler: googleVertexFailedResponseHandler,
122
+ successfulResponseHandler: createJsonResponseHandler(googleVertexEmbedContentResponseSchema),
123
+ abortSignal,
124
+ fetch: this.config.fetch
125
+ });
126
+ return {
127
+ warnings: [],
128
+ embeddings: [response.embedding.values],
129
+ usage: response.usageMetadata?.promptTokenCount == null ? void 0 : { tokens: response.usageMetadata.promptTokenCount },
130
+ response: {
131
+ headers: responseHeaders,
132
+ body: rawValue
133
+ }
134
+ };
135
+ }
136
+ const url = `${this.config.baseURL}/models/${this.modelId}:predict`;
137
+ const { responseHeaders, value: response, rawValue } = await postJsonToApi({
138
+ url,
139
+ headers: mergedHeaders,
140
+ body: {
141
+ instances: values.map((value) => ({
142
+ content: value,
143
+ task_type: googleOptions.taskType,
144
+ title: googleOptions.title
145
+ })),
146
+ parameters: {
147
+ outputDimensionality: googleOptions.outputDimensionality,
148
+ autoTruncate: googleOptions.autoTruncate
149
+ }
150
+ },
151
+ failedResponseHandler: googleVertexFailedResponseHandler,
152
+ successfulResponseHandler: createJsonResponseHandler(googleVertexTextEmbeddingResponseSchema),
153
+ abortSignal,
154
+ fetch: this.config.fetch
155
+ });
156
+ return {
157
+ warnings: [],
158
+ embeddings: response.predictions.map((prediction) => prediction.embeddings.values),
159
+ usage: { tokens: response.predictions.reduce((tokenCount, prediction) => tokenCount + prediction.embeddings.statistics.token_count, 0) },
160
+ response: {
161
+ headers: responseHeaders,
162
+ body: rawValue
163
+ }
164
+ };
165
+ }
233
166
  };
234
- var googleVertexTextEmbeddingResponseSchema = z3.object({
235
- predictions: z3.array(
236
- z3.object({
237
- embeddings: z3.object({
238
- values: z3.array(z3.number()),
239
- statistics: z3.object({
240
- token_count: z3.number()
241
- })
242
- })
243
- })
244
- )
245
- });
246
- var googleVertexEmbedContentResponseSchema = z3.object({
247
- embedding: z3.object({
248
- values: z3.array(z3.number())
249
- }),
250
- usageMetadata: z3.object({
251
- promptTokenCount: z3.number().nullish()
252
- }).nullish()
167
+ const googleVertexTextEmbeddingResponseSchema = z.object({ predictions: z.array(z.object({ embeddings: z.object({
168
+ values: z.array(z.number()),
169
+ statistics: z.object({ token_count: z.number() })
170
+ }) })) });
171
+ const googleVertexEmbedContentResponseSchema = z.object({
172
+ embedding: z.object({ values: z.array(z.number()) }),
173
+ usageMetadata: z.object({ promptTokenCount: z.number().nullish() }).nullish()
253
174
  });
254
175
  function usesEmbedContentEndpoint(modelId) {
255
- return modelId === "gemini-embedding-2" || modelId === "gemini-embedding-2-preview";
176
+ return modelId === "gemini-embedding-2" || modelId === "gemini-embedding-2-preview";
256
177
  }
257
-
258
- // src/google-vertex-image-model.ts
259
- import { GoogleLanguageModel } from "@ai-sdk/google/internal";
260
- import {
261
- convertToBase64,
262
- generateId as defaultGenerateId,
263
- serializeModelOptions as serializeModelOptions2,
264
- WORKFLOW_SERIALIZE as WORKFLOW_SERIALIZE2,
265
- WORKFLOW_DESERIALIZE as WORKFLOW_DESERIALIZE2
266
- } from "@ai-sdk/provider-utils";
267
- var googleVertexImageModelsWithFileInputSupport = /* @__PURE__ */ new Set([
268
- "gemini-2.5-flash-image",
269
- "gemini-3-pro-image-preview",
270
- "gemini-3.1-flash-image-preview"
178
+ //#endregion
179
+ //#region src/google-vertex-image-model.ts
180
+ const googleVertexImageModelsWithFileInputSupport = /* @__PURE__ */ new Set([
181
+ "gemini-2.5-flash-image",
182
+ "gemini-3-pro-image-preview",
183
+ "gemini-3.1-flash-image-preview"
271
184
  ]);
272
- var GoogleVertexImageModel = class _GoogleVertexImageModel {
273
- constructor(modelId, config) {
274
- this.modelId = modelId;
275
- this.config = config;
276
- this.specificationVersion = "v4";
277
- this.maxImagesPerCall = 1;
278
- }
279
- static [WORKFLOW_SERIALIZE2](model) {
280
- return serializeModelOptions2({
281
- modelId: model.modelId,
282
- config: model.config
283
- });
284
- }
285
- static [WORKFLOW_DESERIALIZE2](options) {
286
- return new _GoogleVertexImageModel(options.modelId, options.config);
287
- }
288
- get supportsFileInputs() {
289
- return googleVertexImageModelsWithFileInputSupport.has(this.modelId) ? true : void 0;
290
- }
291
- get supportsMaskInputs() {
292
- return this.supportsFileInputs === true ? false : void 0;
293
- }
294
- get provider() {
295
- return this.config.provider;
296
- }
297
- async doGenerate(options) {
298
- if (!this.modelId.startsWith("gemini-")) {
299
- throw new Error(
300
- "Google image models other than Gemini are no longer supported. Use a model ID that starts with `gemini-`."
301
- );
302
- }
303
- const {
304
- prompt,
305
- size,
306
- aspectRatio,
307
- seed,
308
- providerOptions,
309
- headers,
310
- abortSignal,
311
- files,
312
- mask
313
- } = options;
314
- const warnings = [];
315
- if (mask != null) {
316
- throw new Error(
317
- "Gemini image models do not support mask-based image editing."
318
- );
319
- }
320
- if (size != null) {
321
- warnings.push({
322
- type: "unsupported",
323
- feature: "size",
324
- details: "This model does not support the `size` option. Use `aspectRatio` instead."
325
- });
326
- }
327
- const userContent = [];
328
- if (prompt != null) {
329
- userContent.push({ type: "text", text: prompt });
330
- }
331
- if (files != null && files.length > 0) {
332
- for (const file of files) {
333
- if (file.type === "url") {
334
- userContent.push({
335
- type: "file",
336
- data: { type: "url", url: new URL(file.url) },
337
- mediaType: "image/*"
338
- });
339
- } else {
340
- userContent.push({
341
- type: "file",
342
- data: {
343
- type: "data",
344
- data: typeof file.data === "string" ? file.data : new Uint8Array(file.data)
345
- },
346
- mediaType: file.mediaType
347
- });
348
- }
349
- }
350
- }
351
- const languageModelPrompt = [
352
- { role: "user", content: userContent }
353
- ];
354
- const languageModel = new GoogleLanguageModel(this.modelId, {
355
- provider: this.config.provider,
356
- baseURL: this.config.baseURL,
357
- headers: this.config.headers ?? {},
358
- fetch: this.config.fetch,
359
- generateId: this.config.generateId ?? defaultGenerateId,
360
- supportedUrls: () => ({
361
- "*": [/^https?:\/\/.*$/, /^gs:\/\/.*$/]
362
- })
363
- });
364
- const {
365
- responseModalities: _strippedResponseModalities,
366
- imageConfig: userImageConfig,
367
- ...userVertexOptions
368
- } = providerOptions?.googleVertex ?? providerOptions?.vertex ?? {};
369
- const innerVertexOptions = {
370
- ...userVertexOptions,
371
- responseModalities: ["IMAGE"],
372
- imageConfig: aspectRatio != null || userImageConfig != null ? {
373
- ...userImageConfig,
374
- ...aspectRatio != null ? {
375
- aspectRatio
376
- } : {}
377
- } : void 0
378
- };
379
- const result = await languageModel.doGenerate({
380
- prompt: languageModelPrompt,
381
- seed,
382
- providerOptions: {
383
- googleVertex: innerVertexOptions,
384
- vertex: innerVertexOptions
385
- },
386
- headers,
387
- abortSignal
388
- });
389
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
390
- const images = [];
391
- for (const part of result.content) {
392
- if (part.type === "file" && part.mediaType.startsWith("image/") && part.data.type === "data") {
393
- images.push(convertToBase64(part.data.data));
394
- }
395
- }
396
- const geminiPayload = {
397
- images: images.map(() => ({}))
398
- };
399
- return {
400
- images,
401
- ...result.finishReason.unified === "content-filter" ? { isRetryable: false } : {},
402
- warnings,
403
- providerMetadata: {
404
- googleVertex: geminiPayload,
405
- vertex: geminiPayload
406
- },
407
- response: {
408
- timestamp: currentDate,
409
- modelId: this.modelId,
410
- headers: result.response?.headers
411
- },
412
- usage: result.usage ? {
413
- inputTokens: result.usage.inputTokens.total,
414
- outputTokens: result.usage.outputTokens.total,
415
- totalTokens: (result.usage.inputTokens.total ?? 0) + (result.usage.outputTokens.total ?? 0)
416
- } : void 0
417
- };
418
- }
185
+ var GoogleVertexImageModel = class GoogleVertexImageModel {
186
+ static [WORKFLOW_SERIALIZE](model) {
187
+ return serializeModelOptions({
188
+ modelId: model.modelId,
189
+ config: model.config
190
+ });
191
+ }
192
+ static [WORKFLOW_DESERIALIZE](options) {
193
+ return new GoogleVertexImageModel(options.modelId, options.config);
194
+ }
195
+ get supportsFileInputs() {
196
+ return googleVertexImageModelsWithFileInputSupport.has(this.modelId) ? true : void 0;
197
+ }
198
+ get supportsMaskInputs() {
199
+ return this.supportsFileInputs === true ? false : void 0;
200
+ }
201
+ get provider() {
202
+ return this.config.provider;
203
+ }
204
+ constructor(modelId, config) {
205
+ this.modelId = modelId;
206
+ this.config = config;
207
+ this.specificationVersion = "v4";
208
+ this.maxImagesPerCall = 1;
209
+ }
210
+ async doGenerate(options) {
211
+ if (!this.modelId.startsWith("gemini-")) throw new Error("Google image models other than Gemini are no longer supported. Use a model ID that starts with `gemini-`.");
212
+ const { prompt, size, aspectRatio, seed, providerOptions, headers, abortSignal, files, mask } = options;
213
+ const warnings = [];
214
+ if (mask != null) throw new Error("Gemini image models do not support mask-based image editing.");
215
+ if (size != null) warnings.push({
216
+ type: "unsupported",
217
+ feature: "size",
218
+ details: "This model does not support the `size` option. Use `aspectRatio` instead."
219
+ });
220
+ const userContent = [];
221
+ if (prompt != null) userContent.push({
222
+ type: "text",
223
+ text: prompt
224
+ });
225
+ if (files != null && files.length > 0) for (const file of files) if (file.type === "url") userContent.push({
226
+ type: "file",
227
+ data: {
228
+ type: "url",
229
+ url: new URL(file.url)
230
+ },
231
+ mediaType: "image/*"
232
+ });
233
+ else userContent.push({
234
+ type: "file",
235
+ data: {
236
+ type: "data",
237
+ data: typeof file.data === "string" ? file.data : new Uint8Array(file.data)
238
+ },
239
+ mediaType: file.mediaType
240
+ });
241
+ const languageModelPrompt = [{
242
+ role: "user",
243
+ content: userContent
244
+ }];
245
+ const languageModel = new GoogleLanguageModel(this.modelId, {
246
+ provider: this.config.provider,
247
+ baseURL: this.config.baseURL,
248
+ headers: this.config.headers ?? {},
249
+ fetch: this.config.fetch,
250
+ generateId: this.config.generateId ?? generateId,
251
+ supportedUrls: () => ({ "*": [/^https?:\/\/.*$/, /^gs:\/\/.*$/] })
252
+ });
253
+ const { responseModalities: _strippedResponseModalities, imageConfig: userImageConfig, ...userVertexOptions } = providerOptions?.googleVertex ?? providerOptions?.vertex ?? {};
254
+ const innerVertexOptions = {
255
+ ...userVertexOptions,
256
+ responseModalities: ["IMAGE"],
257
+ imageConfig: aspectRatio != null || userImageConfig != null ? {
258
+ ...userImageConfig,
259
+ ...aspectRatio != null ? { aspectRatio } : {}
260
+ } : void 0
261
+ };
262
+ const result = await languageModel.doGenerate({
263
+ prompt: languageModelPrompt,
264
+ seed,
265
+ providerOptions: {
266
+ googleVertex: innerVertexOptions,
267
+ vertex: innerVertexOptions
268
+ },
269
+ headers,
270
+ abortSignal
271
+ });
272
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
273
+ const images = [];
274
+ for (const part of result.content) if (part.type === "file" && part.mediaType.startsWith("image/") && part.data.type === "data") images.push(convertToBase64(part.data.data));
275
+ const geminiPayload = { images: images.map(() => ({})) };
276
+ return {
277
+ images,
278
+ ...result.finishReason.unified === "content-filter" ? { isRetryable: false } : {},
279
+ warnings,
280
+ providerMetadata: {
281
+ googleVertex: geminiPayload,
282
+ vertex: geminiPayload
283
+ },
284
+ response: {
285
+ timestamp: currentDate,
286
+ modelId: this.modelId,
287
+ headers: result.response?.headers
288
+ },
289
+ usage: result.usage ? {
290
+ inputTokens: result.usage.inputTokens.total,
291
+ outputTokens: result.usage.outputTokens.total,
292
+ totalTokens: (result.usage.inputTokens.total ?? 0) + (result.usage.outputTokens.total ?? 0)
293
+ } : void 0
294
+ };
295
+ }
419
296
  };
420
-
421
- // src/google-vertex-cloud-tts-speech-model.ts
422
- import {
423
- combineHeaders as combineHeaders2,
424
- convertBase64ToUint8Array,
425
- createJsonResponseHandler as createJsonResponseHandler2,
426
- postJsonToApi as postJsonToApi2,
427
- resolve as resolve2,
428
- serializeModelOptions as serializeModelOptions3,
429
- WORKFLOW_DESERIALIZE as WORKFLOW_DESERIALIZE3,
430
- WORKFLOW_SERIALIZE as WORKFLOW_SERIALIZE3
431
- } from "@ai-sdk/provider-utils";
432
- import { z as z4 } from "zod/v4";
433
- var DEFAULT_VOICE = "Kore";
434
- var DEFAULT_LANGUAGE = "en-US";
435
- var CHIRP3_HD_VOICE_INFIX = "Chirp3-HD";
436
- var CLOUD_TTS_SYNTHESIZE_URL = "https://texttospeech.googleapis.com/v1/text:synthesize";
437
- var GoogleVertexCloudTTSSpeechModel = class _GoogleVertexCloudTTSSpeechModel {
438
- constructor(modelId, config) {
439
- this.modelId = modelId;
440
- this.config = config;
441
- this.specificationVersion = "v4";
442
- }
443
- static [WORKFLOW_SERIALIZE3](model) {
444
- return serializeModelOptions3({
445
- modelId: model.modelId,
446
- config: model.config
447
- });
448
- }
449
- static [WORKFLOW_DESERIALIZE3](options) {
450
- return new _GoogleVertexCloudTTSSpeechModel(options.modelId, options.config);
451
- }
452
- get provider() {
453
- return this.config.provider;
454
- }
455
- async doGenerate(options) {
456
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
457
- const warnings = [];
458
- const {
459
- text,
460
- voice = DEFAULT_VOICE,
461
- outputFormat,
462
- instructions,
463
- speed,
464
- language
465
- } = options;
466
- let voiceName;
467
- let languageCode;
468
- if (voice.includes(CHIRP3_HD_VOICE_INFIX)) {
469
- voiceName = voice;
470
- const localePrefix = voice.split(CHIRP3_HD_VOICE_INFIX)[0].replace(/-$/, "");
471
- languageCode = language ?? (localePrefix || DEFAULT_LANGUAGE);
472
- } else {
473
- languageCode = language ?? DEFAULT_LANGUAGE;
474
- voiceName = `${languageCode}-${CHIRP3_HD_VOICE_INFIX}-${voice}`;
475
- }
476
- if (instructions != null) {
477
- warnings.push({
478
- type: "unsupported",
479
- feature: "instructions",
480
- details: "Google Cloud Text-to-Speech Chirp 3: HD voices do not support the `instructions` option. It was ignored."
481
- });
482
- }
483
- if (outputFormat != null && outputFormat !== "wav") {
484
- warnings.push({
485
- type: "unsupported",
486
- feature: "outputFormat",
487
- details: `Unsupported output format: ${outputFormat}. Using wav instead.`
488
- });
489
- }
490
- const requestBody = {
491
- input: { text },
492
- voice: { languageCode, name: voiceName },
493
- audioConfig: {
494
- audioEncoding: "LINEAR16",
495
- ...speed != null ? { speakingRate: speed } : {}
496
- }
497
- };
498
- const {
499
- value: response,
500
- responseHeaders,
501
- rawValue: rawResponse
502
- } = await postJsonToApi2({
503
- url: CLOUD_TTS_SYNTHESIZE_URL,
504
- headers: combineHeaders2(
505
- this.config.headers ? await resolve2(this.config.headers) : void 0,
506
- options.headers
507
- ),
508
- body: requestBody,
509
- failedResponseHandler: googleVertexFailedResponseHandler,
510
- successfulResponseHandler: createJsonResponseHandler2(
511
- googleVertexCloudTTSResponseSchema
512
- ),
513
- abortSignal: options.abortSignal,
514
- fetch: this.config.fetch
515
- });
516
- const audio = response.audioContent != null ? convertBase64ToUint8Array(response.audioContent) : new Uint8Array(0);
517
- return {
518
- audio,
519
- warnings,
520
- request: {
521
- body: JSON.stringify(requestBody)
522
- },
523
- response: {
524
- timestamp: currentDate,
525
- modelId: this.modelId,
526
- headers: responseHeaders,
527
- body: rawResponse
528
- },
529
- providerMetadata: {
530
- google: {
531
- mimeType: "audio/wav"
532
- }
533
- }
534
- };
535
- }
297
+ //#endregion
298
+ //#region src/google-vertex-cloud-tts-speech-model.ts
299
+ const DEFAULT_VOICE = "Kore";
300
+ const DEFAULT_LANGUAGE = "en-US";
301
+ const CHIRP3_HD_VOICE_INFIX = "Chirp3-HD";
302
+ const CLOUD_TTS_SYNTHESIZE_URL = "https://texttospeech.googleapis.com/v1/text:synthesize";
303
+ /**
304
+ * Speech model for Chirp 3: HD voices on the Google Cloud Text-to-Speech API.
305
+ *
306
+ * Unlike the Gemini TTS models (which go through the Vertex
307
+ * `generateContent` endpoint via `GoogleSpeechModel`), Chirp 3: HD voices are
308
+ * served by the dedicated Cloud Text-to-Speech `text:synthesize` endpoint,
309
+ * reusing the provider's Google Cloud credentials.
310
+ */
311
+ var GoogleVertexCloudTTSSpeechModel = class GoogleVertexCloudTTSSpeechModel {
312
+ static [WORKFLOW_SERIALIZE](model) {
313
+ return serializeModelOptions({
314
+ modelId: model.modelId,
315
+ config: model.config
316
+ });
317
+ }
318
+ static [WORKFLOW_DESERIALIZE](options) {
319
+ return new GoogleVertexCloudTTSSpeechModel(options.modelId, options.config);
320
+ }
321
+ get provider() {
322
+ return this.config.provider;
323
+ }
324
+ constructor(modelId, config) {
325
+ this.modelId = modelId;
326
+ this.config = config;
327
+ this.specificationVersion = "v4";
328
+ }
329
+ async doGenerate(options) {
330
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
331
+ const warnings = [];
332
+ const { text, voice = DEFAULT_VOICE, outputFormat, instructions, speed, language } = options;
333
+ let voiceName;
334
+ let languageCode;
335
+ if (voice.includes(CHIRP3_HD_VOICE_INFIX)) {
336
+ voiceName = voice;
337
+ const localePrefix = voice.split(CHIRP3_HD_VOICE_INFIX)[0].replace(/-$/, "");
338
+ languageCode = language ?? (localePrefix || DEFAULT_LANGUAGE);
339
+ } else {
340
+ languageCode = language ?? DEFAULT_LANGUAGE;
341
+ voiceName = `${languageCode}-${CHIRP3_HD_VOICE_INFIX}-${voice}`;
342
+ }
343
+ if (instructions != null) warnings.push({
344
+ type: "unsupported",
345
+ feature: "instructions",
346
+ details: "Google Cloud Text-to-Speech Chirp 3: HD voices do not support the `instructions` option. It was ignored."
347
+ });
348
+ if (outputFormat != null && outputFormat !== "wav") warnings.push({
349
+ type: "unsupported",
350
+ feature: "outputFormat",
351
+ details: `Unsupported output format: ${outputFormat}. Using wav instead.`
352
+ });
353
+ const requestBody = {
354
+ input: { text },
355
+ voice: {
356
+ languageCode,
357
+ name: voiceName
358
+ },
359
+ audioConfig: {
360
+ audioEncoding: "LINEAR16",
361
+ ...speed != null ? { speakingRate: speed } : {}
362
+ }
363
+ };
364
+ const { value: response, responseHeaders, rawValue: rawResponse } = await postJsonToApi({
365
+ url: CLOUD_TTS_SYNTHESIZE_URL,
366
+ headers: combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, options.headers),
367
+ body: requestBody,
368
+ failedResponseHandler: googleVertexFailedResponseHandler,
369
+ successfulResponseHandler: createJsonResponseHandler(googleVertexCloudTTSResponseSchema),
370
+ abortSignal: options.abortSignal,
371
+ fetch: this.config.fetch
372
+ });
373
+ return {
374
+ audio: response.audioContent != null ? convertBase64ToUint8Array(response.audioContent) : /* @__PURE__ */ new Uint8Array(0),
375
+ warnings,
376
+ request: { body: JSON.stringify(requestBody) },
377
+ response: {
378
+ timestamp: currentDate,
379
+ modelId: this.modelId,
380
+ headers: responseHeaders,
381
+ body: rawResponse
382
+ },
383
+ providerMetadata: { google: { mimeType: "audio/wav" } }
384
+ };
385
+ }
536
386
  };
537
- var googleVertexCloudTTSResponseSchema = z4.object({
538
- audioContent: z4.string().nullish()
539
- });
540
-
541
- // src/google-vertex-tools.ts
542
- import { googleTools } from "@ai-sdk/google/internal";
543
- var googleVertexTools = {
544
- googleSearch: googleTools.googleSearch,
545
- enterpriseWebSearch: googleTools.enterpriseWebSearch,
546
- googleMaps: googleTools.googleMaps,
547
- urlContext: googleTools.urlContext,
548
- fileSearch: googleTools.fileSearch,
549
- codeExecution: googleTools.codeExecution,
550
- vertexRagStore: googleTools.vertexRagStore
387
+ const googleVertexCloudTTSResponseSchema = z.object({ audioContent: z.string().nullish() });
388
+ //#endregion
389
+ //#region src/google-vertex-tools.ts
390
+ const googleVertexTools = {
391
+ googleSearch: googleTools.googleSearch,
392
+ enterpriseWebSearch: googleTools.enterpriseWebSearch,
393
+ googleMaps: googleTools.googleMaps,
394
+ urlContext: googleTools.urlContext,
395
+ fileSearch: googleTools.fileSearch,
396
+ codeExecution: googleTools.codeExecution,
397
+ vertexRagStore: googleTools.vertexRagStore
551
398
  };
552
-
553
- // src/google-vertex-transcription-model.ts
554
- import {
555
- combineHeaders as combineHeaders3,
556
- convertUint8ArrayToBase64,
557
- createJsonResponseHandler as createJsonResponseHandler3,
558
- parseProviderOptions as parseProviderOptions2,
559
- postJsonToApi as postJsonToApi3,
560
- resolve as resolve3,
561
- serializeModelOptions as serializeModelOptions4,
562
- WORKFLOW_DESERIALIZE as WORKFLOW_DESERIALIZE4,
563
- WORKFLOW_SERIALIZE as WORKFLOW_SERIALIZE4
564
- } from "@ai-sdk/provider-utils";
565
- import { z as z6 } from "zod/v4";
566
-
567
- // src/google-vertex-transcription-model-options.ts
568
- import { z as z5 } from "zod/v4";
569
- var googleVertexTranscriptionProviderOptionsSchema = z5.object({
570
- /**
571
- * BCP-47 language codes to recognize (e.g. `['en-US']`), or `['auto']` to let
572
- * Chirp auto-detect the spoken language. Defaults to `['auto']`. For
573
- * `telephony`, pass a supported explicit language code.
574
- */
575
- languageCodes: z5.array(z5.string()).optional(),
576
- /**
577
- * Whether to add punctuation to the transcript. Defaults to `true`.
578
- */
579
- enableAutomaticPunctuation: z5.boolean().optional(),
580
- /**
581
- * Whether to include word-level timestamps. Defaults to `true` so the
582
- * transcription result can include segments.
583
- *
584
- * Enabling word-level timestamps can reduce transcription quality and speed
585
- * for Chirp models.
586
- */
587
- enableWordTimeOffsets: z5.boolean().optional(),
588
- /**
589
- * The Cloud Speech-to-Text region for the request (e.g. `'us'`, `'eu'`,
590
- * `'us-central1'`). Defaults to the provider `location`.
591
- *
592
- * Note: Speech-to-Text regions differ from Vertex AI regions. Chirp is only
593
- * available in specific Speech-to-Text regions and is not available in the
594
- * `global` location.
595
- */
596
- region: z5.string().optional()
399
+ //#endregion
400
+ //#region src/google-vertex-transcription-model-options.ts
401
+ const googleVertexTranscriptionProviderOptionsSchema = z.object({
402
+ /**
403
+ * BCP-47 language codes to recognize (e.g. `['en-US']`), or `['auto']` to let
404
+ * Chirp auto-detect the spoken language. Defaults to `['auto']`. For
405
+ * `telephony`, pass a supported explicit language code.
406
+ */
407
+ languageCodes: z.array(z.string()).optional(),
408
+ /**
409
+ * Whether to add punctuation to the transcript. Defaults to `true`.
410
+ */
411
+ enableAutomaticPunctuation: z.boolean().optional(),
412
+ /**
413
+ * Whether to include word-level timestamps. Defaults to `true` so the
414
+ * transcription result can include segments.
415
+ *
416
+ * Enabling word-level timestamps can reduce transcription quality and speed
417
+ * for Chirp models.
418
+ */
419
+ enableWordTimeOffsets: z.boolean().optional(),
420
+ /**
421
+ * The Cloud Speech-to-Text region for the request (e.g. `'us'`, `'eu'`,
422
+ * `'us-central1'`). Defaults to the provider `location`. Must be a single
423
+ * DNS label (letters, digits, and hyphens).
424
+ *
425
+ * Note: Speech-to-Text regions differ from Vertex AI regions. Chirp is only
426
+ * available in specific Speech-to-Text regions and is not available in the
427
+ * `global` location.
428
+ */
429
+ region: z.string().optional()
597
430
  });
598
-
599
- // src/google-vertex-transcription-model.ts
431
+ //#endregion
432
+ //#region src/google-vertex-transcription-model.ts
600
433
  function parseDurationSeconds(value) {
601
- if (value == null) {
602
- return void 0;
603
- }
604
- const seconds = Number.parseFloat(value);
605
- return Number.isFinite(seconds) ? seconds : void 0;
434
+ if (value == null) return;
435
+ const seconds = Number.parseFloat(value);
436
+ return Number.isFinite(seconds) ? seconds : void 0;
606
437
  }
607
438
  function convertBcp47ToIso6391(value) {
608
- if (value == null) {
609
- return void 0;
610
- }
611
- try {
612
- const language = new Intl.Locale(value).language;
613
- return language.length === 2 ? language : void 0;
614
- } catch {
615
- return void 0;
616
- }
439
+ if (value == null) return;
440
+ try {
441
+ const language = new Intl.Locale(value).language;
442
+ return language.length === 2 ? language : void 0;
443
+ } catch {
444
+ return;
445
+ }
617
446
  }
618
- var GoogleVertexTranscriptionModel = class _GoogleVertexTranscriptionModel {
619
- constructor(modelId, config) {
620
- this.modelId = modelId;
621
- this.config = config;
622
- this.specificationVersion = "v4";
623
- }
624
- static [WORKFLOW_SERIALIZE4](model) {
625
- return serializeModelOptions4({
626
- modelId: model.modelId,
627
- config: model.config
628
- });
629
- }
630
- static [WORKFLOW_DESERIALIZE4](options) {
631
- return new _GoogleVertexTranscriptionModel(options.modelId, options.config);
632
- }
633
- get provider() {
634
- return this.config.provider;
635
- }
636
- async doGenerate(options) {
637
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
638
- const warnings = [];
639
- let googleOptions;
640
- for (const provider of ["googleVertex", "vertex", "google"]) {
641
- googleOptions = await parseProviderOptions2({
642
- provider,
643
- providerOptions: options.providerOptions,
644
- schema: googleVertexTranscriptionProviderOptionsSchema
645
- });
646
- if (googleOptions != null) {
647
- break;
648
- }
649
- }
650
- const region = googleOptions?.region ?? this.config.location;
651
- const languageCodes = googleOptions?.languageCodes ?? ["auto"];
652
- const content = typeof options.audio === "string" ? options.audio : convertUint8ArrayToBase64(options.audio);
653
- const requestBody = {
654
- config: {
655
- model: this.modelId,
656
- languageCodes,
657
- // Let Speech-to-Text auto-detect the audio encoding (wav/mp3/flac/…).
658
- autoDecodingConfig: {},
659
- features: {
660
- // Word timing populates `segments`.
661
- enableWordTimeOffsets: googleOptions?.enableWordTimeOffsets ?? true,
662
- enableAutomaticPunctuation: googleOptions?.enableAutomaticPunctuation ?? true
663
- }
664
- },
665
- content
666
- };
667
- const host = region === "global" ? "speech.googleapis.com" : `${region}-speech.googleapis.com`;
668
- const url = `https://${host}/v2/projects/${this.config.project}/locations/${region}/recognizers/_:recognize`;
669
- const {
670
- value: response,
671
- responseHeaders,
672
- rawValue: rawResponse
673
- } = await postJsonToApi3({
674
- url,
675
- headers: combineHeaders3(
676
- this.config.headers ? await resolve3(this.config.headers) : void 0,
677
- options.headers
678
- ),
679
- body: requestBody,
680
- failedResponseHandler: googleVertexFailedResponseHandler,
681
- successfulResponseHandler: createJsonResponseHandler3(
682
- googleVertexTranscriptionResponseSchema
683
- ),
684
- abortSignal: options.abortSignal,
685
- fetch: this.config.fetch
686
- });
687
- const results = response.results ?? [];
688
- const text = results.map((result) => result.alternatives?.[0]?.transcript ?? "").join(" ").trim();
689
- const segments = results.flatMap(
690
- (result) => result.alternatives?.[0]?.words?.flatMap((word) => {
691
- const startSecond = parseDurationSeconds(word.startOffset);
692
- const endSecond = parseDurationSeconds(word.endOffset);
693
- return word.word == null || startSecond == null || endSecond == null ? [] : [{ text: word.word, startSecond, endSecond }];
694
- }) ?? []
695
- );
696
- const language = convertBcp47ToIso6391(results[0]?.languageCode);
697
- return {
698
- text,
699
- segments,
700
- language,
701
- durationInSeconds: parseDurationSeconds(
702
- response.metadata?.totalBilledDuration
703
- ),
704
- warnings,
705
- response: {
706
- timestamp: currentDate,
707
- modelId: this.modelId,
708
- headers: responseHeaders,
709
- body: rawResponse
710
- }
711
- };
712
- }
447
+ var GoogleVertexTranscriptionModel = class GoogleVertexTranscriptionModel {
448
+ static [WORKFLOW_SERIALIZE](model) {
449
+ return serializeModelOptions({
450
+ modelId: model.modelId,
451
+ config: model.config
452
+ });
453
+ }
454
+ static [WORKFLOW_DESERIALIZE](options) {
455
+ return new GoogleVertexTranscriptionModel(options.modelId, options.config);
456
+ }
457
+ get provider() {
458
+ return this.config.provider;
459
+ }
460
+ constructor(modelId, config) {
461
+ this.modelId = modelId;
462
+ this.config = config;
463
+ this.specificationVersion = "v4";
464
+ }
465
+ async doGenerate(options) {
466
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
467
+ const warnings = [];
468
+ let googleOptions;
469
+ for (const provider of [
470
+ "googleVertex",
471
+ "vertex",
472
+ "google"
473
+ ]) {
474
+ googleOptions = await parseProviderOptions({
475
+ provider,
476
+ providerOptions: options.providerOptions,
477
+ schema: googleVertexTranscriptionProviderOptionsSchema
478
+ });
479
+ if (googleOptions != null) break;
480
+ }
481
+ const region = googleOptions?.region ?? this.config.location;
482
+ if (!isValidHostnamePart(region)) throw new InvalidArgumentError({
483
+ argument: googleOptions?.region != null ? "region" : "location",
484
+ message: "Invalid Google Cloud Speech-to-Text region. Expected a single DNS label (letters, digits, and hyphens)."
485
+ });
486
+ const languageCodes = googleOptions?.languageCodes ?? ["auto"];
487
+ const content = typeof options.audio === "string" ? options.audio : convertUint8ArrayToBase64(options.audio);
488
+ const requestBody = {
489
+ config: {
490
+ model: this.modelId,
491
+ languageCodes,
492
+ autoDecodingConfig: {},
493
+ features: {
494
+ enableWordTimeOffsets: googleOptions?.enableWordTimeOffsets ?? true,
495
+ enableAutomaticPunctuation: googleOptions?.enableAutomaticPunctuation ?? true
496
+ }
497
+ },
498
+ content
499
+ };
500
+ const url = `https://${region === "global" ? "speech.googleapis.com" : `${region}-speech.googleapis.com`}/v2/projects/${this.config.project}/locations/${region}/recognizers/_:recognize`;
501
+ const { value: response, responseHeaders, rawValue: rawResponse } = await postJsonToApi({
502
+ url,
503
+ headers: combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, options.headers),
504
+ body: requestBody,
505
+ failedResponseHandler: googleVertexFailedResponseHandler,
506
+ successfulResponseHandler: createJsonResponseHandler(googleVertexTranscriptionResponseSchema),
507
+ abortSignal: options.abortSignal,
508
+ fetch: this.config.fetch
509
+ });
510
+ const results = response.results ?? [];
511
+ return {
512
+ text: results.map((result) => result.alternatives?.[0]?.transcript ?? "").join(" ").trim(),
513
+ segments: results.flatMap((result) => result.alternatives?.[0]?.words?.flatMap((word) => {
514
+ const startSecond = parseDurationSeconds(word.startOffset);
515
+ const endSecond = parseDurationSeconds(word.endOffset);
516
+ return word.word == null || startSecond == null || endSecond == null ? [] : [{
517
+ text: word.word,
518
+ startSecond,
519
+ endSecond
520
+ }];
521
+ }) ?? []),
522
+ language: convertBcp47ToIso6391(results[0]?.languageCode),
523
+ durationInSeconds: parseDurationSeconds(response.metadata?.totalBilledDuration),
524
+ warnings,
525
+ response: {
526
+ timestamp: currentDate,
527
+ modelId: this.modelId,
528
+ headers: responseHeaders,
529
+ body: rawResponse
530
+ }
531
+ };
532
+ }
713
533
  };
714
- var googleVertexTranscriptionResponseSchema = z6.object({
715
- results: z6.array(
716
- z6.object({
717
- alternatives: z6.array(
718
- z6.object({
719
- transcript: z6.string().nullish(),
720
- words: z6.array(
721
- z6.object({
722
- word: z6.string().nullish(),
723
- startOffset: z6.string().nullish(),
724
- endOffset: z6.string().nullish()
725
- })
726
- ).nullish()
727
- })
728
- ).nullish(),
729
- languageCode: z6.string().nullish()
730
- })
731
- ).nullish(),
732
- metadata: z6.object({
733
- totalBilledDuration: z6.string().nullish()
734
- }).nullish()
534
+ const googleVertexTranscriptionResponseSchema = z.object({
535
+ results: z.array(z.object({
536
+ alternatives: z.array(z.object({
537
+ transcript: z.string().nullish(),
538
+ words: z.array(z.object({
539
+ word: z.string().nullish(),
540
+ startOffset: z.string().nullish(),
541
+ endOffset: z.string().nullish()
542
+ })).nullish()
543
+ })).nullish(),
544
+ languageCode: z.string().nullish()
545
+ })).nullish(),
546
+ metadata: z.object({ totalBilledDuration: z.string().nullish() }).nullish()
735
547
  });
736
-
737
- // src/gemini-transcription/google-vertex-gemini-transcription-model.ts
738
- import {
739
- InvalidArgumentError
740
- } from "@ai-sdk/provider";
741
- import {
742
- combineHeaders as combineHeaders4,
743
- connectToWebSocket,
744
- convertToBase64 as convertToBase642,
745
- createJsonResponseHandler as createJsonResponseHandler4,
746
- parseProviderOptions as parseProviderOptions3,
747
- postJsonToApi as postJsonToApi4,
748
- resolve as resolve4,
749
- safeParseJSON,
750
- serializeModelOptions as serializeModelOptions5,
751
- waitForWebSocketBufferDrain,
752
- WORKFLOW_DESERIALIZE as WORKFLOW_DESERIALIZE5,
753
- WORKFLOW_SERIALIZE as WORKFLOW_SERIALIZE5
754
- } from "@ai-sdk/provider-utils";
755
- import { z as z8 } from "zod/v4";
756
-
757
- // src/gemini-transcription/google-vertex-gemini-transcription-model-options.ts
758
- import { z as z7 } from "zod/v4";
759
- var googleVertexGeminiTranscriptionModelOptions = z7.object({
760
- /**
761
- * BCP-47 language codes providing hints about the languages present in the
762
- * audio. If omitted or empty, defaults to automatic language detection.
763
- */
764
- languageCodes: z7.array(z7.string()).optional(),
765
- /**
766
- * Custom vocabulary phrases, which bias the speech recognition model
767
- * toward recognizing specific terms.
768
- */
769
- customVocabulary: z7.array(z7.string()).optional(),
770
- /**
771
- * Enables word-level timestamp generation.
772
- */
773
- wordTimestamp: z7.boolean().optional(),
774
- /**
775
- * Enables speaker diarization.
776
- */
777
- diarization: z7.boolean().optional(),
778
- /**
779
- * Transcription output formatting mode.
780
- *
781
- * - `VERBATIM` (default): exact literal transcript preserving filler
782
- * words, repetitions, and false starts.
783
- * - `SMART`: cleans up and structures the transcript in real time —
784
- * disfluency removal, inline self-corrections, structured formatting
785
- * (lists, numbers, dates, paragraph breaks), and grammar/casing polish.
786
- */
787
- mode: z7.enum(["SMART", "VERBATIM"]).optional()
548
+ //#endregion
549
+ //#region src/gemini-transcription/google-vertex-gemini-transcription-model-options.ts
550
+ /**
551
+ * Speech recognition options for Gemini transcription models on Vertex,
552
+ * shared by unary (`gemini-3.5-transcribe`) and live
553
+ * (`gemini-3.5-transcribe-live`) variants. Maps onto Google's
554
+ * `AudioTranscriptionConfig`.
555
+ */
556
+ const googleVertexGeminiTranscriptionModelOptions = z.object({
557
+ /**
558
+ * BCP-47 language codes providing hints about the languages present in the
559
+ * audio. If omitted or empty, defaults to automatic language detection.
560
+ */
561
+ languageCodes: z.array(z.string()).optional(),
562
+ /**
563
+ * Custom vocabulary phrases, which bias the speech recognition model
564
+ * toward recognizing specific terms.
565
+ */
566
+ customVocabulary: z.array(z.string()).optional(),
567
+ /**
568
+ * Enables word-level timestamp generation.
569
+ */
570
+ wordTimestamp: z.boolean().optional(),
571
+ /**
572
+ * Enables speaker diarization.
573
+ */
574
+ diarization: z.boolean().optional(),
575
+ /**
576
+ * Transcription output formatting mode.
577
+ *
578
+ * - `VERBATIM` (default): exact literal transcript preserving filler
579
+ * words, repetitions, and false starts.
580
+ * - `SMART`: cleans up and structures the transcript in real time —
581
+ * disfluency removal, inline self-corrections, structured formatting
582
+ * (lists, numbers, dates, paragraph breaks), and grammar/casing polish.
583
+ */
584
+ mode: z.enum(["SMART", "VERBATIM"]).optional()
788
585
  });
789
-
790
- // src/gemini-transcription/google-vertex-gemini-transcription-model.ts
791
- var liveWebSocketPath = "google.cloud.aiplatform.v1.LlmBidiService/BidiGenerateContent";
792
- var defaultFinishGraceMs = 3e3;
586
+ //#endregion
587
+ //#region src/gemini-transcription/google-vertex-gemini-transcription-model.ts
588
+ const liveWebSocketPath = "google.cloud.aiplatform.v1.LlmBidiService/BidiGenerateContent";
589
+ /**
590
+ * After the input audio has ended, finish when no terminal signal
591
+ * (`turnComplete` / idle `interactionStatus`) arrives within this window.
592
+ * Trailing transcripts reset the timer.
593
+ */
594
+ const defaultFinishGraceMs = 3e3;
595
+ /** Live transcription is only supported by `*-live` model variants. */
793
596
  function isLiveTranscriptionModelId(modelId) {
794
- return modelId.includes("-live");
597
+ return modelId.includes("-live");
795
598
  }
599
+ /** Regional Vertex hostname (mirrors the provider's base-URL host rules). */
796
600
  function vertexHost(location) {
797
- if (location === "global") return "aiplatform.googleapis.com";
798
- if (location === "eu" || location === "us") {
799
- return `aiplatform.${location}.rep.googleapis.com`;
800
- }
801
- return `${location}-aiplatform.googleapis.com`;
601
+ if (location === "global") return "aiplatform.googleapis.com";
602
+ if (location === "eu" || location === "us") return `aiplatform.${location}.rep.googleapis.com`;
603
+ return `${location}-aiplatform.googleapis.com`;
802
604
  }
803
- var GoogleVertexGeminiTranscriptionModel = class _GoogleVertexGeminiTranscriptionModel {
804
- constructor(modelId, config) {
805
- this.modelId = modelId;
806
- this.config = config;
807
- this.specificationVersion = "v4";
808
- }
809
- static [WORKFLOW_SERIALIZE5](model) {
810
- return serializeModelOptions5({
811
- modelId: model.modelId,
812
- config: model.config
813
- });
814
- }
815
- static [WORKFLOW_DESERIALIZE5](options) {
816
- return new _GoogleVertexGeminiTranscriptionModel(
817
- options.modelId,
818
- options.config
819
- );
820
- }
821
- get provider() {
822
- return this.config.provider;
823
- }
824
- async parseOptions(providerOptions) {
825
- for (const provider of ["googleVertex", "vertex", "google"]) {
826
- const parsed = await parseProviderOptions3({
827
- provider,
828
- providerOptions,
829
- schema: googleVertexGeminiTranscriptionModelOptions
830
- });
831
- if (parsed != null) return parsed;
832
- }
833
- }
834
- async doGenerate(options) {
835
- if (isLiveTranscriptionModelId(this.modelId)) {
836
- throw new InvalidArgumentError({
837
- argument: "modelId",
838
- message: `Model '${this.modelId}' only supports streaming transcription. Use experimental_streamTranscribe, or a unary model such as 'gemini-3.5-transcribe'.`
839
- });
840
- }
841
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
842
- const warnings = [];
843
- const googleOptions = await this.parseOptions(options.providerOptions);
844
- const audioTranscriptionConfig = buildAudioTranscriptionConfig(googleOptions);
845
- const requestBody = {
846
- contents: [
847
- {
848
- role: "user",
849
- parts: [
850
- {
851
- inlineData: {
852
- mimeType: options.mediaType,
853
- data: convertToBase642(options.audio)
854
- }
855
- }
856
- ]
857
- }
858
- ],
859
- ...audioTranscriptionConfig != null ? { generationConfig: { audioTranscriptionConfig } } : {}
860
- };
861
- const {
862
- value: response,
863
- responseHeaders,
864
- rawValue: rawResponse
865
- } = await postJsonToApi4({
866
- url: `${this.config.baseURL}/models/${this.modelId}:generateContent`,
867
- headers: combineHeaders4(
868
- this.config.headers ? await resolve4(this.config.headers) : void 0,
869
- options.headers
870
- ),
871
- body: requestBody,
872
- failedResponseHandler: googleVertexFailedResponseHandler,
873
- successfulResponseHandler: createJsonResponseHandler4(
874
- googleVertexGeminiTranscriptionResponseSchema
875
- ),
876
- abortSignal: options.abortSignal,
877
- fetch: this.config.fetch
878
- });
879
- const parts = response.candidates?.[0]?.content?.parts ?? [];
880
- const plainText = parts.map((part) => part.text ?? "").join("");
881
- const transcriptionText = parts.map((part) => part.audioTranscription?.text ?? "").join("");
882
- const text = plainText !== "" ? plainText : transcriptionText;
883
- let language;
884
- const segments = [];
885
- for (const part of parts) {
886
- const transcription = part.audioTranscription;
887
- if (transcription == null) continue;
888
- language ??= transcription.languageCode ?? void 0;
889
- for (const word of transcription.words ?? []) {
890
- const startSecond = parseOffsetSeconds(word.startOffset);
891
- const endSecond = parseOffsetSeconds(word.endOffset);
892
- if (word.word == null || startSecond == null || endSecond == null) {
893
- continue;
894
- }
895
- segments.push({ text: word.word, startSecond, endSecond });
896
- }
897
- }
898
- return {
899
- text,
900
- segments,
901
- language,
902
- durationInSeconds: void 0,
903
- warnings,
904
- response: {
905
- timestamp: currentDate,
906
- modelId: this.modelId,
907
- headers: responseHeaders,
908
- body: rawResponse
909
- },
910
- ...response.usageMetadata != null ? {
911
- providerMetadata: {
912
- google: { usageMetadata: response.usageMetadata }
913
- }
914
- } : {}
915
- };
916
- }
917
- async doStream(options) {
918
- if (!isLiveTranscriptionModelId(this.modelId)) {
919
- throw new InvalidArgumentError({
920
- argument: "modelId",
921
- message: `Model '${this.modelId}' does not support streaming transcription. Use a live model such as 'gemini-3.5-transcribe-live'.`
922
- });
923
- }
924
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
925
- const warnings = [];
926
- const googleOptions = await this.parseOptions(options.providerOptions);
927
- validateLiveInputAudioFormat(options.inputAudioFormat);
928
- const headers = combineHeaders4(
929
- this.config.headers ? await resolve4(this.config.headers) : void 0,
930
- options.headers
931
- );
932
- const { project, location } = this.config;
933
- const modelResource = `projects/${project}/locations/${location}/publishers/google/models/${this.modelId}`;
934
- const url = new URL(
935
- `wss://${vertexHost(location)}/ws/${liveWebSocketPath}`
936
- );
937
- const setup = {
938
- model: modelResource,
939
- inputAudioTranscription: buildAudioTranscriptionConfig(googleOptions) ?? {}
940
- };
941
- return {
942
- request: { body: setup },
943
- response: {
944
- timestamp: currentDate,
945
- modelId: this.modelId
946
- },
947
- stream: createVertexLiveTranscriptionStream({
948
- webSocket: this.config.webSocket,
949
- url,
950
- headers,
951
- setup,
952
- inputAudioRate: options.inputAudioFormat.rate ?? 16e3,
953
- finishGraceMs: this.config._internal?.finishGraceMs ?? defaultFinishGraceMs,
954
- warnings,
955
- audio: options.audio,
956
- abortSignal: options.abortSignal,
957
- includeRawChunks: options.includeRawChunks
958
- })
959
- };
960
- }
605
+ /**
606
+ * Gemini transcription on Vertex AI. Unary variants transcribe via
607
+ * `generateContent`; live variants stream over the Vertex Live API WebSocket
608
+ * (`LlmBidiService/BidiGenerateContent`) with OAuth Bearer authentication
609
+ * from the provider's resolved headers.
610
+ */
611
+ var GoogleVertexGeminiTranscriptionModel = class GoogleVertexGeminiTranscriptionModel {
612
+ static [WORKFLOW_SERIALIZE](model) {
613
+ return serializeModelOptions({
614
+ modelId: model.modelId,
615
+ config: model.config
616
+ });
617
+ }
618
+ static [WORKFLOW_DESERIALIZE](options) {
619
+ return new GoogleVertexGeminiTranscriptionModel(options.modelId, options.config);
620
+ }
621
+ get provider() {
622
+ return this.config.provider;
623
+ }
624
+ constructor(modelId, config) {
625
+ this.modelId = modelId;
626
+ this.config = config;
627
+ this.specificationVersion = "v4";
628
+ }
629
+ async parseOptions(providerOptions) {
630
+ for (const provider of [
631
+ "googleVertex",
632
+ "vertex",
633
+ "google"
634
+ ]) {
635
+ const parsed = await parseProviderOptions({
636
+ provider,
637
+ providerOptions,
638
+ schema: googleVertexGeminiTranscriptionModelOptions
639
+ });
640
+ if (parsed != null) return parsed;
641
+ }
642
+ }
643
+ async doGenerate(options) {
644
+ if (isLiveTranscriptionModelId(this.modelId)) throw new InvalidArgumentError({
645
+ argument: "modelId",
646
+ message: `Model '${this.modelId}' only supports streaming transcription. Use experimental_streamTranscribe, or a unary model such as 'gemini-3.5-transcribe'.`
647
+ });
648
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
649
+ const warnings = [];
650
+ const audioTranscriptionConfig = buildAudioTranscriptionConfig(await this.parseOptions(options.providerOptions));
651
+ const requestBody = {
652
+ contents: [{
653
+ role: "user",
654
+ parts: [{ inlineData: {
655
+ mimeType: options.mediaType,
656
+ data: convertToBase64(options.audio)
657
+ } }]
658
+ }],
659
+ ...audioTranscriptionConfig != null ? { generationConfig: { audioTranscriptionConfig } } : {}
660
+ };
661
+ const { value: response, responseHeaders, rawValue: rawResponse } = await postJsonToApi({
662
+ url: `${this.config.baseURL}/models/${this.modelId}:generateContent`,
663
+ headers: combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, options.headers),
664
+ body: requestBody,
665
+ failedResponseHandler: googleVertexFailedResponseHandler,
666
+ successfulResponseHandler: createJsonResponseHandler(googleVertexGeminiTranscriptionResponseSchema),
667
+ abortSignal: options.abortSignal,
668
+ fetch: this.config.fetch
669
+ });
670
+ const parts = response.candidates?.[0]?.content?.parts ?? [];
671
+ const plainText = parts.map((part) => part.text ?? "").join("");
672
+ const transcriptionText = parts.map((part) => part.audioTranscription?.text ?? "").join("");
673
+ const text = plainText !== "" ? plainText : transcriptionText;
674
+ let language;
675
+ const segments = [];
676
+ for (const part of parts) {
677
+ const transcription = part.audioTranscription;
678
+ if (transcription == null) continue;
679
+ language ??= transcription.languageCode ?? void 0;
680
+ for (const word of transcription.words ?? []) {
681
+ const startSecond = parseOffsetSeconds(word.startOffset);
682
+ const endSecond = parseOffsetSeconds(word.endOffset);
683
+ if (word.word == null || startSecond == null || endSecond == null) continue;
684
+ segments.push({
685
+ text: word.word,
686
+ startSecond,
687
+ endSecond
688
+ });
689
+ }
690
+ }
691
+ return {
692
+ text,
693
+ segments,
694
+ language,
695
+ durationInSeconds: void 0,
696
+ warnings,
697
+ response: {
698
+ timestamp: currentDate,
699
+ modelId: this.modelId,
700
+ headers: responseHeaders,
701
+ body: rawResponse
702
+ },
703
+ ...response.usageMetadata != null ? {
704
+ usage: response.usageMetadata,
705
+ providerMetadata: { google: { usageMetadata: response.usageMetadata } }
706
+ } : {}
707
+ };
708
+ }
709
+ async doStream(options) {
710
+ if (!isLiveTranscriptionModelId(this.modelId)) throw new InvalidArgumentError({
711
+ argument: "modelId",
712
+ message: `Model '${this.modelId}' does not support streaming transcription. Use a live model such as 'gemini-3.5-transcribe-live'.`
713
+ });
714
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
715
+ const warnings = [];
716
+ const googleOptions = await this.parseOptions(options.providerOptions);
717
+ validateLiveInputAudioFormat(options.inputAudioFormat);
718
+ const headers = combineHeaders(this.config.headers ? await resolve(this.config.headers) : void 0, options.headers);
719
+ const { project, location } = this.config;
720
+ const modelResource = `projects/${project}/locations/${location}/publishers/google/models/${this.modelId}`;
721
+ const url = new URL(`wss://${vertexHost(location)}/ws/${liveWebSocketPath}`);
722
+ const setup = {
723
+ model: modelResource,
724
+ inputAudioTranscription: buildAudioTranscriptionConfig(googleOptions) ?? {}
725
+ };
726
+ return {
727
+ request: { body: setup },
728
+ response: {
729
+ timestamp: currentDate,
730
+ modelId: this.modelId
731
+ },
732
+ stream: createVertexLiveTranscriptionStream({
733
+ webSocket: this.config.webSocket,
734
+ url,
735
+ headers,
736
+ setup,
737
+ inputAudioRate: options.inputAudioFormat.rate ?? 16e3,
738
+ finishGraceMs: this.config._internal?.finishGraceMs ?? defaultFinishGraceMs,
739
+ warnings,
740
+ audio: options.audio,
741
+ abortSignal: options.abortSignal,
742
+ includeRawChunks: options.includeRawChunks
743
+ })
744
+ };
745
+ }
961
746
  };
962
- function createVertexLiveTranscriptionStream({
963
- webSocket,
964
- url,
965
- headers,
966
- setup,
967
- inputAudioRate,
968
- finishGraceMs,
969
- warnings,
970
- audio,
971
- abortSignal,
972
- includeRawChunks
973
- }) {
974
- let finished = false;
975
- let cleanup = () => {
976
- };
977
- return new ReadableStream({
978
- start: (controller) => {
979
- let audioReader;
980
- let connection;
981
- let resolveSetupComplete;
982
- const setupComplete = new Promise((resolvePromise) => {
983
- resolveSetupComplete = resolvePromise;
984
- });
985
- let segmentCounter = 0;
986
- let segmentBuffer = "";
987
- let fullText = "";
988
- let latestInterim = "";
989
- let language;
990
- let audioEnded = false;
991
- let usageMetadata;
992
- let finishTimer;
993
- const segmentId = () => `google-segment-${segmentCounter}`;
994
- const cancelPendingFinish = () => {
995
- if (finishTimer != null) {
996
- clearTimeout(finishTimer);
997
- finishTimer = void 0;
998
- }
999
- };
1000
- const schedulePendingFinish = () => {
1001
- if (finished || !audioEnded) return;
1002
- cancelPendingFinish();
1003
- finishTimer = setTimeout(() => {
1004
- finishTimer = void 0;
1005
- finish();
1006
- }, finishGraceMs);
1007
- };
1008
- cleanup = (closeCode) => {
1009
- cancelPendingFinish();
1010
- if (audioReader != null) {
1011
- void audioReader.cancel().catch(() => {
1012
- });
1013
- } else {
1014
- void audio.cancel().catch(() => {
1015
- });
1016
- }
1017
- connection?.close(closeCode);
1018
- };
1019
- const finishWithError = (error) => {
1020
- if (finished) return;
1021
- finished = true;
1022
- cleanup();
1023
- controller.error(error);
1024
- };
1025
- const completeSegment = () => {
1026
- if (segmentBuffer === "") {
1027
- if (latestInterim === "") return;
1028
- segmentBuffer = latestInterim;
1029
- }
1030
- latestInterim = "";
1031
- controller.enqueue({
1032
- type: "transcript-final",
1033
- id: segmentId(),
1034
- text: segmentBuffer
1035
- });
1036
- fullText += fullText === "" ? segmentBuffer : ` ${segmentBuffer}`;
1037
- segmentBuffer = "";
1038
- segmentCounter++;
1039
- };
1040
- const finish = () => {
1041
- if (finished) return;
1042
- completeSegment();
1043
- finished = true;
1044
- controller.enqueue({
1045
- type: "finish",
1046
- text: fullText,
1047
- segments: [],
1048
- language,
1049
- durationInSeconds: void 0,
1050
- ...usageMetadata != null ? { providerMetadata: { google: { usageMetadata } } } : {}
1051
- });
1052
- controller.close();
1053
- cleanup(1e3);
1054
- };
1055
- const sendAudio = async (socket) => {
1056
- audioReader = audio.getReader();
1057
- try {
1058
- while (true) {
1059
- const { done, value } = await audioReader.read();
1060
- if (done || finished) break;
1061
- socket.send(
1062
- JSON.stringify({
1063
- realtimeInput: {
1064
- audio: {
1065
- data: convertToBase642(value),
1066
- mimeType: `audio/pcm;rate=${inputAudioRate}`
1067
- }
1068
- }
1069
- })
1070
- );
1071
- await waitForWebSocketBufferDrain(socket);
1072
- }
1073
- } finally {
1074
- audioReader.releaseLock();
1075
- audioReader = void 0;
1076
- }
1077
- if (!finished) {
1078
- socket.send(
1079
- JSON.stringify({ realtimeInput: { audioStreamEnd: true } })
1080
- );
1081
- audioEnded = true;
1082
- schedulePendingFinish();
1083
- }
1084
- };
1085
- connection = connectToWebSocket({
1086
- url,
1087
- headers,
1088
- webSocket,
1089
- abortSignal,
1090
- onAbort: finishWithError,
1091
- onProcessingError: finishWithError,
1092
- onOpen: (socket) => {
1093
- controller.enqueue({ type: "stream-start", warnings });
1094
- socket.send(JSON.stringify({ setup }));
1095
- void setupComplete.then(() => finished ? void 0 : sendAudio(socket)).catch(finishWithError);
1096
- },
1097
- onMessageText: async (text) => {
1098
- if (finished) return;
1099
- const parsed = await safeParseJSON({ text });
1100
- if (!parsed.success) return;
1101
- const message = parsed.value;
1102
- if (includeRawChunks) {
1103
- controller.enqueue({ type: "raw", rawValue: message });
1104
- }
1105
- if (message.setupComplete != null) {
1106
- resolveSetupComplete();
1107
- }
1108
- if (message.usageMetadata != null) {
1109
- usageMetadata = message.usageMetadata;
1110
- }
1111
- if (message.error != null) {
1112
- finishWithError(
1113
- new Error(message.error.message ?? "Vertex Live API error")
1114
- );
1115
- return;
1116
- }
1117
- const serverContent = message.serverContent;
1118
- const interim = serverContent?.interimInputTranscription;
1119
- if (interim?.text) {
1120
- schedulePendingFinish();
1121
- latestInterim = interim.text;
1122
- controller.enqueue({
1123
- type: "transcript-partial",
1124
- id: segmentId(),
1125
- text: interim.text
1126
- });
1127
- }
1128
- const transcription = serverContent?.inputTranscription ?? message.inputTranscription;
1129
- if (transcription != null) {
1130
- if (transcription.languageCode != null) {
1131
- language = transcription.languageCode;
1132
- }
1133
- if (transcription.text) {
1134
- schedulePendingFinish();
1135
- latestInterim = "";
1136
- segmentBuffer += transcription.text;
1137
- controller.enqueue({
1138
- type: "transcript-delta",
1139
- id: segmentId(),
1140
- delta: transcription.text
1141
- });
1142
- }
1143
- if (transcription.finished === true) {
1144
- completeSegment();
1145
- }
1146
- }
1147
- if (serverContent?.turnComplete) {
1148
- completeSegment();
1149
- }
1150
- const interactionStatus = serverContent?.interactionStatus;
1151
- if (audioEnded && (interactionStatus === "IDLE" || interactionStatus === "REQUIRES_ACTION" || serverContent?.turnComplete === true && interactionStatus == null)) {
1152
- finish();
1153
- }
1154
- },
1155
- onSocketError: () => {
1156
- finishWithError(
1157
- new Error(
1158
- "Vertex Live transcription error." + (webSocket == null ? " Note: the native WebSocket implementation cannot send the Authorization header required by Vertex. Pass a header-capable WebSocket implementation (e.g. the 'ws' package) via createVertex({ webSocket })." : "")
1159
- )
1160
- );
1161
- },
1162
- onClose: ({ code, reason }) => {
1163
- if (finished) return;
1164
- if (audioEnded) {
1165
- finish();
1166
- return;
1167
- }
1168
- finishWithError(
1169
- new Error(
1170
- `Vertex Live transcription WebSocket closed unexpectedly before finishing (code ${code ?? "unknown"}${reason ? `, reason: ${reason}` : ""}).`
1171
- )
1172
- );
1173
- }
1174
- });
1175
- },
1176
- cancel: () => {
1177
- if (finished) return;
1178
- finished = true;
1179
- cleanup();
1180
- }
1181
- });
747
+ function createVertexLiveTranscriptionStream({ webSocket, url, headers, setup, inputAudioRate, finishGraceMs, warnings, audio, abortSignal, includeRawChunks }) {
748
+ let finished = false;
749
+ let cleanup = () => {};
750
+ return new ReadableStream({
751
+ start: (controller) => {
752
+ let audioReader;
753
+ let connection;
754
+ let resolveSetupComplete;
755
+ const setupComplete = new Promise((resolvePromise) => {
756
+ resolveSetupComplete = resolvePromise;
757
+ });
758
+ let segmentCounter = 0;
759
+ let segmentBuffer = "";
760
+ let fullText = "";
761
+ let latestInterim = "";
762
+ let language;
763
+ let audioEnded = false;
764
+ let usageMetadata;
765
+ let finishTimer;
766
+ const segmentId = () => `google-segment-${segmentCounter}`;
767
+ const cancelPendingFinish = () => {
768
+ if (finishTimer != null) {
769
+ clearTimeout(finishTimer);
770
+ finishTimer = void 0;
771
+ }
772
+ };
773
+ const schedulePendingFinish = () => {
774
+ if (finished || !audioEnded) return;
775
+ cancelPendingFinish();
776
+ finishTimer = setTimeout(() => {
777
+ finishTimer = void 0;
778
+ finish();
779
+ }, finishGraceMs);
780
+ };
781
+ cleanup = (closeCode) => {
782
+ cancelPendingFinish();
783
+ if (audioReader != null) audioReader.cancel().catch(() => {});
784
+ else audio.cancel().catch(() => {});
785
+ connection?.close(closeCode);
786
+ };
787
+ const finishWithError = (error) => {
788
+ if (finished) return;
789
+ finished = true;
790
+ cleanup();
791
+ controller.error(error);
792
+ };
793
+ const completeSegment = () => {
794
+ if (segmentBuffer === "") {
795
+ if (latestInterim === "") return;
796
+ segmentBuffer = latestInterim;
797
+ }
798
+ latestInterim = "";
799
+ controller.enqueue({
800
+ type: "transcript-final",
801
+ id: segmentId(),
802
+ text: segmentBuffer
803
+ });
804
+ fullText += fullText === "" ? segmentBuffer : ` ${segmentBuffer}`;
805
+ segmentBuffer = "";
806
+ segmentCounter++;
807
+ };
808
+ const finish = () => {
809
+ if (finished) return;
810
+ completeSegment();
811
+ finished = true;
812
+ controller.enqueue({
813
+ type: "finish",
814
+ text: fullText,
815
+ segments: [],
816
+ language,
817
+ durationInSeconds: void 0,
818
+ ...usageMetadata != null ? {
819
+ usage: usageMetadata,
820
+ providerMetadata: { google: { usageMetadata } }
821
+ } : {}
822
+ });
823
+ controller.close();
824
+ cleanup(1e3);
825
+ };
826
+ const sendAudio = async (socket) => {
827
+ audioReader = audio.getReader();
828
+ try {
829
+ while (true) {
830
+ const { done, value } = await audioReader.read();
831
+ if (done || finished) break;
832
+ socket.send(JSON.stringify({ realtimeInput: { audio: {
833
+ data: convertToBase64(value),
834
+ mimeType: `audio/pcm;rate=${inputAudioRate}`
835
+ } } }));
836
+ await waitForWebSocketBufferDrain(socket);
837
+ }
838
+ } finally {
839
+ audioReader.releaseLock();
840
+ audioReader = void 0;
841
+ }
842
+ if (!finished) {
843
+ socket.send(JSON.stringify({ realtimeInput: { audioStreamEnd: true } }));
844
+ audioEnded = true;
845
+ schedulePendingFinish();
846
+ }
847
+ };
848
+ connection = connectToWebSocket({
849
+ url,
850
+ headers,
851
+ webSocket,
852
+ abortSignal,
853
+ onAbort: finishWithError,
854
+ onProcessingError: finishWithError,
855
+ onOpen: (socket) => {
856
+ controller.enqueue({
857
+ type: "stream-start",
858
+ warnings
859
+ });
860
+ socket.send(JSON.stringify({ setup }));
861
+ setupComplete.then(() => finished ? void 0 : sendAudio(socket)).catch(finishWithError);
862
+ },
863
+ onMessageText: async (text) => {
864
+ if (finished) return;
865
+ const parsed = await safeParseJSON({ text });
866
+ if (!parsed.success) return;
867
+ const message = parsed.value;
868
+ if (includeRawChunks) controller.enqueue({
869
+ type: "raw",
870
+ rawValue: message
871
+ });
872
+ if (message.setupComplete != null) resolveSetupComplete();
873
+ if (message.usageMetadata != null) usageMetadata = message.usageMetadata;
874
+ if (message.error != null) {
875
+ finishWithError(new Error(message.error.message ?? "Vertex Live API error"));
876
+ return;
877
+ }
878
+ const serverContent = message.serverContent;
879
+ const interim = serverContent?.interimInputTranscription;
880
+ if (interim?.text) {
881
+ schedulePendingFinish();
882
+ latestInterim = interim.text;
883
+ controller.enqueue({
884
+ type: "transcript-partial",
885
+ id: segmentId(),
886
+ text: interim.text
887
+ });
888
+ }
889
+ const transcription = serverContent?.inputTranscription ?? message.inputTranscription;
890
+ if (transcription != null) {
891
+ if (transcription.languageCode != null) language = transcription.languageCode;
892
+ if (transcription.text) {
893
+ schedulePendingFinish();
894
+ latestInterim = "";
895
+ segmentBuffer += transcription.text;
896
+ controller.enqueue({
897
+ type: "transcript-delta",
898
+ id: segmentId(),
899
+ delta: transcription.text
900
+ });
901
+ }
902
+ if (transcription.finished === true) completeSegment();
903
+ }
904
+ if (serverContent?.turnComplete) completeSegment();
905
+ const interactionStatus = serverContent?.interactionStatus;
906
+ if (audioEnded && (interactionStatus === "IDLE" || interactionStatus === "REQUIRES_ACTION" || serverContent?.turnComplete === true && interactionStatus == null)) finish();
907
+ },
908
+ onSocketError: () => {
909
+ finishWithError(/* @__PURE__ */ new Error("Vertex Live transcription error." + (webSocket == null ? " Note: the native WebSocket implementation cannot send the Authorization header required by Vertex. Pass a header-capable WebSocket implementation (e.g. the 'ws' package) via createVertex({ webSocket })." : "")));
910
+ },
911
+ onClose: ({ code, reason }) => {
912
+ if (finished) return;
913
+ if (audioEnded) {
914
+ finish();
915
+ return;
916
+ }
917
+ finishWithError(/* @__PURE__ */ new Error(`Vertex Live transcription WebSocket closed unexpectedly before finishing (code ${code ?? "unknown"}${reason ? `, reason: ${reason}` : ""}).`));
918
+ }
919
+ });
920
+ },
921
+ cancel: () => {
922
+ if (finished) return;
923
+ finished = true;
924
+ cleanup();
925
+ }
926
+ });
1182
927
  }
928
+ /**
929
+ * Builds Google's `AudioTranscriptionConfig` from provider options; returns
930
+ * undefined when no options are set.
931
+ */
1183
932
  function buildAudioTranscriptionConfig(options) {
1184
- if (options == null) return void 0;
1185
- const config = {};
1186
- if (options.languageCodes != null) {
1187
- config.languageCodes = options.languageCodes;
1188
- }
1189
- if (options.customVocabulary != null) {
1190
- config.customVocabulary = options.customVocabulary;
1191
- }
1192
- if (options.wordTimestamp != null) {
1193
- config.wordTimestamp = options.wordTimestamp;
1194
- }
1195
- if (options.diarization != null) {
1196
- config.diarization = options.diarization;
1197
- }
1198
- if (options.mode != null) {
1199
- config.mode = options.mode;
1200
- }
1201
- return Object.keys(config).length > 0 ? config : void 0;
933
+ if (options == null) return void 0;
934
+ const config = {};
935
+ if (options.languageCodes != null) config.languageCodes = options.languageCodes;
936
+ if (options.customVocabulary != null) config.customVocabulary = options.customVocabulary;
937
+ if (options.wordTimestamp != null) config.wordTimestamp = options.wordTimestamp;
938
+ if (options.diarization != null) config.diarization = options.diarization;
939
+ if (options.mode != null) config.mode = options.mode;
940
+ return Object.keys(config).length > 0 ? config : void 0;
1202
941
  }
1203
942
  function validateLiveInputAudioFormat(inputAudioFormat) {
1204
- if (inputAudioFormat.type !== "audio/pcm" || inputAudioFormat.rate != null && inputAudioFormat.rate !== 16e3) {
1205
- throw new InvalidArgumentError({
1206
- argument: "inputAudioFormat",
1207
- message: "The Gemini Live transcription API only supports 16kHz 16-bit PCM input audio."
1208
- });
1209
- }
943
+ if (inputAudioFormat.type !== "audio/pcm" || inputAudioFormat.rate != null && inputAudioFormat.rate !== 16e3) throw new InvalidArgumentError({
944
+ argument: "inputAudioFormat",
945
+ message: "The Gemini Live transcription API only supports 16kHz 16-bit PCM input audio."
946
+ });
1210
947
  }
948
+ /** Parses a Google duration offset such as `"1s"` or `"9.400s"` to seconds. */
1211
949
  function parseOffsetSeconds(offset) {
1212
- if (offset == null) return void 0;
1213
- const parsed = Number.parseFloat(offset);
1214
- return Number.isFinite(parsed) ? parsed : void 0;
950
+ if (offset == null) return void 0;
951
+ const parsed = Number.parseFloat(offset);
952
+ return Number.isFinite(parsed) ? parsed : void 0;
1215
953
  }
1216
- var googleVertexGeminiTranscriptionWordSchema = z8.object({
1217
- word: z8.string().nullish(),
1218
- startOffset: z8.string().nullish(),
1219
- endOffset: z8.string().nullish()
954
+ const googleVertexGeminiTranscriptionWordSchema = z.object({
955
+ word: z.string().nullish(),
956
+ startOffset: z.string().nullish(),
957
+ endOffset: z.string().nullish()
1220
958
  });
1221
- var googleVertexGeminiTranscriptionResponseSchema = z8.object({
1222
- candidates: z8.array(
1223
- z8.object({
1224
- content: z8.object({
1225
- parts: z8.array(
1226
- z8.object({
1227
- text: z8.string().nullish(),
1228
- audioTranscription: z8.object({
1229
- text: z8.string().nullish(),
1230
- languageCode: z8.string().nullish(),
1231
- speakerLabel: z8.string().nullish(),
1232
- words: z8.array(googleVertexGeminiTranscriptionWordSchema).nullish()
1233
- }).nullish()
1234
- })
1235
- ).nullish()
1236
- }).nullish()
1237
- })
1238
- ).nullish(),
1239
- usageMetadata: z8.record(z8.string(), z8.unknown()).nullish()
959
+ const googleVertexGeminiTranscriptionResponseSchema = z.object({
960
+ candidates: z.array(z.object({ content: z.object({ parts: z.array(z.object({
961
+ text: z.string().nullish(),
962
+ audioTranscription: z.object({
963
+ text: z.string().nullish(),
964
+ languageCode: z.string().nullish(),
965
+ speakerLabel: z.string().nullish(),
966
+ words: z.array(googleVertexGeminiTranscriptionWordSchema).nullish()
967
+ }).nullish()
968
+ })).nullish() }).nullish() })).nullish(),
969
+ usageMetadata: z.record(z.string(), z.unknown()).nullish()
1240
970
  });
1241
-
1242
- // src/google-vertex-video-model.ts
1243
- import {
1244
- AISDKError
1245
- } from "@ai-sdk/provider";
1246
- import {
1247
- combineHeaders as combineHeaders5,
1248
- convertUint8ArrayToBase64 as convertUint8ArrayToBase642,
1249
- createJsonResponseHandler as createJsonResponseHandler5,
1250
- parseProviderOptions as parseProviderOptions4,
1251
- postJsonToApi as postJsonToApi5,
1252
- resolve as resolve5
1253
- } from "@ai-sdk/provider-utils";
1254
- import { z as z10 } from "zod/v4";
1255
-
1256
- // src/google-vertex-video-model-options.ts
1257
- import { lazySchema, zodSchema } from "@ai-sdk/provider-utils";
1258
- import { z as z9 } from "zod/v4";
1259
- var googleVertexVideoModelOptionsSchema = lazySchema(
1260
- () => zodSchema(
1261
- z9.looseObject({
1262
- pollIntervalMs: z9.number().positive().nullish(),
1263
- pollTimeoutMs: z9.number().positive().nullish(),
1264
- personGeneration: z9.enum(["dont_allow", "allow_adult", "allow_all"]).nullish(),
1265
- negativePrompt: z9.string().nullish(),
1266
- generateAudio: z9.boolean().nullish(),
1267
- gcsOutputDirectory: z9.string().nullish(),
1268
- referenceImages: z9.array(
1269
- z9.object({
1270
- bytesBase64Encoded: z9.string().nullish(),
1271
- gcsUri: z9.string().nullish()
1272
- })
1273
- ).nullish()
1274
- })
1275
- )
1276
- );
1277
-
1278
- // src/google-vertex-video-model.ts
971
+ //#endregion
972
+ //#region src/google-vertex-video-model-options.ts
973
+ const googleVertexVideoModelOptionsSchema = lazySchema(() => zodSchema(z.looseObject({
974
+ pollIntervalMs: z.number().positive().nullish(),
975
+ pollTimeoutMs: z.number().positive().nullish(),
976
+ personGeneration: z.enum([
977
+ "dont_allow",
978
+ "allow_adult",
979
+ "allow_all"
980
+ ]).nullish(),
981
+ negativePrompt: z.string().nullish(),
982
+ generateAudio: z.boolean().nullish(),
983
+ gcsOutputDirectory: z.string().nullish(),
984
+ referenceImages: z.array(z.object({
985
+ bytesBase64Encoded: z.string().nullish(),
986
+ gcsUri: z.string().nullish()
987
+ })).nullish()
988
+ })));
989
+ //#endregion
990
+ //#region src/google-vertex-video-model.ts
1279
991
  function getFirstFrameImage(options) {
1280
- return options.frameImages?.find((frame) => frame.frameType === "first_frame")?.image;
992
+ return options.frameImages?.find((frame) => frame.frameType === "first_frame")?.image;
1281
993
  }
1282
994
  function resolveStartImage(options) {
1283
- return getFirstFrameImage(options) ?? options.image;
995
+ return getFirstFrameImage(options) ?? options.image;
1284
996
  }
1285
997
  function getLastFrameImage(options) {
1286
- return options.frameImages?.find((frame) => frame.frameType === "last_frame")?.image;
998
+ return options.frameImages?.find((frame) => frame.frameType === "last_frame")?.image;
1287
999
  }
1288
1000
  function getInputReferences(options) {
1289
- if (options.frameImages != null && options.frameImages.length > 0) {
1290
- return void 0;
1291
- }
1292
- return options.inputReferences != null && options.inputReferences.length > 0 ? options.inputReferences : void 0;
1001
+ if (options.frameImages != null && options.frameImages.length > 0) return;
1002
+ return options.inputReferences != null && options.inputReferences.length > 0 ? options.inputReferences : void 0;
1293
1003
  }
1294
1004
  function convertFileToVertexImage(file, warnings) {
1295
- if (file.type === "url") {
1296
- if (file.url.startsWith("gs://")) {
1297
- return {
1298
- gcsUri: file.url,
1299
- mimeType: "image/png"
1300
- };
1301
- }
1302
- warnings.push({
1303
- type: "unsupported",
1304
- feature: "URL-based image input",
1305
- details: "Vertex AI video models require base64-encoded images or GCS URIs. URL will be ignored."
1306
- });
1307
- return void 0;
1308
- }
1309
- const base64Data = typeof file.data === "string" ? file.data : convertUint8ArrayToBase642(file.data);
1310
- return {
1311
- bytesBase64Encoded: base64Data,
1312
- mimeType: file.mediaType || "image/png"
1313
- };
1005
+ if (file.type === "url") {
1006
+ if (file.url.startsWith("gs://")) return {
1007
+ gcsUri: file.url,
1008
+ mimeType: "image/png"
1009
+ };
1010
+ warnings.push({
1011
+ type: "unsupported",
1012
+ feature: "URL-based image input",
1013
+ details: "Vertex AI video models require base64-encoded images or GCS URIs. URL will be ignored."
1014
+ });
1015
+ return;
1016
+ }
1017
+ return {
1018
+ bytesBase64Encoded: typeof file.data === "string" ? file.data : convertUint8ArrayToBase64(file.data),
1019
+ mimeType: file.mediaType || "image/png"
1020
+ };
1314
1021
  }
1315
1022
  function convertInputReferenceImage(file, warnings) {
1316
- const image = convertFileToVertexImage(file, warnings);
1317
- return image != null ? { image, referenceType: "asset" } : void 0;
1023
+ const image = convertFileToVertexImage(file, warnings);
1024
+ return image != null ? {
1025
+ image,
1026
+ referenceType: "asset"
1027
+ } : void 0;
1318
1028
  }
1319
1029
  var GoogleVertexVideoModel = class {
1320
- constructor(modelId, config) {
1321
- this.modelId = modelId;
1322
- this.config = config;
1323
- this.specificationVersion = "v4";
1324
- }
1325
- get provider() {
1326
- return this.config.provider;
1327
- }
1328
- get maxVideosPerCall() {
1329
- return 4;
1330
- }
1331
- async buildRequest(options) {
1332
- const warnings = [];
1333
- const googleVertexOptions = await parseProviderOptions4({
1334
- provider: "googleVertex",
1335
- providerOptions: options.providerOptions,
1336
- schema: googleVertexVideoModelOptionsSchema
1337
- }) ?? await parseProviderOptions4({
1338
- provider: "vertex",
1339
- providerOptions: options.providerOptions,
1340
- schema: googleVertexVideoModelOptionsSchema
1341
- });
1342
- const instances = [{}];
1343
- const instance = instances[0];
1344
- if (options.prompt != null) {
1345
- instance.prompt = options.prompt;
1346
- }
1347
- const startImage = resolveStartImage(options);
1348
- if (startImage != null) {
1349
- const image = convertFileToVertexImage(startImage, warnings);
1350
- if (image != null) {
1351
- instance.image = image;
1352
- }
1353
- }
1354
- const lastFrameImage = getLastFrameImage(options);
1355
- if (lastFrameImage != null) {
1356
- const lastFrame = convertFileToVertexImage(lastFrameImage, warnings);
1357
- if (lastFrame != null) {
1358
- instance.lastFrame = lastFrame;
1359
- }
1360
- }
1361
- const inputReferences = getInputReferences(options);
1362
- if (inputReferences != null) {
1363
- instance.referenceImages = inputReferences.flatMap((reference) => {
1364
- const converted = convertInputReferenceImage(reference, warnings);
1365
- return converted != null ? [converted] : [];
1366
- });
1367
- } else if (googleVertexOptions?.referenceImages != null) {
1368
- instance.referenceImages = googleVertexOptions.referenceImages;
1369
- }
1370
- const parameters = {
1371
- sampleCount: options.n
1372
- };
1373
- if (options.aspectRatio) {
1374
- parameters.aspectRatio = options.aspectRatio;
1375
- }
1376
- if (options.resolution) {
1377
- const resolutionMap = {
1378
- "1280x720": "720p",
1379
- "1920x1080": "1080p",
1380
- "3840x2160": "4k"
1381
- };
1382
- parameters.resolution = resolutionMap[options.resolution] || options.resolution;
1383
- }
1384
- if (options.duration) {
1385
- parameters.durationSeconds = options.duration;
1386
- }
1387
- if (options.seed) {
1388
- parameters.seed = options.seed;
1389
- }
1390
- const generateAudio = options.generateAudio ?? googleVertexOptions?.generateAudio;
1391
- if (generateAudio != null) {
1392
- parameters.generateAudio = generateAudio;
1393
- }
1394
- if (googleVertexOptions != null) {
1395
- const opts = googleVertexOptions;
1396
- if (opts.personGeneration !== void 0 && opts.personGeneration !== null) {
1397
- parameters.personGeneration = opts.personGeneration;
1398
- }
1399
- if (opts.negativePrompt !== void 0 && opts.negativePrompt !== null) {
1400
- parameters.negativePrompt = opts.negativePrompt;
1401
- }
1402
- if (opts.gcsOutputDirectory !== void 0 && opts.gcsOutputDirectory !== null) {
1403
- parameters.gcsOutputDirectory = opts.gcsOutputDirectory;
1404
- }
1405
- for (const [key, value] of Object.entries(opts)) {
1406
- if (![
1407
- "pollIntervalMs",
1408
- "pollTimeoutMs",
1409
- "personGeneration",
1410
- "negativePrompt",
1411
- "generateAudio",
1412
- "gcsOutputDirectory",
1413
- "referenceImages"
1414
- ].includes(key)) {
1415
- parameters[key] = value;
1416
- }
1417
- }
1418
- }
1419
- return { instances, parameters, warnings, googleVertexOptions };
1420
- }
1421
- buildCompletedResult({
1422
- finalOperation,
1423
- responseHeaders,
1424
- warnings,
1425
- currentDate
1426
- }) {
1427
- const response = finalOperation.response;
1428
- if (!response?.videos || response.videos.length === 0) {
1429
- throw new AISDKError({
1430
- name: "VERTEX_VIDEO_GENERATION_ERROR",
1431
- message: `No videos in response. Response: ${JSON.stringify(finalOperation)}`
1432
- });
1433
- }
1434
- const videos = [];
1435
- const videoMetadata = [];
1436
- for (const video of response.videos) {
1437
- if (video.bytesBase64Encoded) {
1438
- videos.push({
1439
- type: "base64",
1440
- data: video.bytesBase64Encoded,
1441
- mediaType: video.mimeType || "video/mp4"
1442
- });
1443
- videoMetadata.push({
1444
- mimeType: video.mimeType
1445
- });
1446
- } else if (video.gcsUri) {
1447
- videos.push({
1448
- type: "url",
1449
- url: video.gcsUri,
1450
- mediaType: video.mimeType || "video/mp4"
1451
- });
1452
- videoMetadata.push({
1453
- gcsUri: video.gcsUri,
1454
- mimeType: video.mimeType
1455
- });
1456
- }
1457
- }
1458
- if (videos.length === 0) {
1459
- throw new AISDKError({
1460
- name: "VERTEX_VIDEO_GENERATION_ERROR",
1461
- message: "No valid videos in response"
1462
- });
1463
- }
1464
- return {
1465
- status: "completed",
1466
- videos,
1467
- warnings,
1468
- response: {
1469
- timestamp: currentDate,
1470
- modelId: this.modelId,
1471
- headers: responseHeaders
1472
- },
1473
- providerMetadata: /* @__PURE__ */ (() => {
1474
- const payload = { videos: videoMetadata };
1475
- return {
1476
- googleVertex: payload,
1477
- // Legacy keys preserved for backward compatibility.
1478
- "google-vertex": payload,
1479
- vertex: payload
1480
- };
1481
- })()
1482
- };
1483
- }
1484
- async doStart(options) {
1485
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
1486
- const { instances, parameters, warnings } = await this.buildRequest(options);
1487
- const { value: operation, responseHeaders } = await postJsonToApi5({
1488
- url: `${this.config.baseURL}/models/${this.modelId}:predictLongRunning`,
1489
- headers: combineHeaders5(
1490
- await resolve5(this.config.headers),
1491
- options.headers
1492
- ),
1493
- body: {
1494
- instances,
1495
- parameters
1496
- },
1497
- successfulResponseHandler: createJsonResponseHandler5(
1498
- googleVertexOperationSchema
1499
- ),
1500
- failedResponseHandler: googleVertexFailedResponseHandler,
1501
- abortSignal: options.abortSignal,
1502
- fetch: this.config.fetch
1503
- });
1504
- const operationName = operation.name;
1505
- if (!operationName) {
1506
- throw new AISDKError({
1507
- name: "VERTEX_VIDEO_GENERATION_ERROR",
1508
- message: "No operation name returned from API"
1509
- });
1510
- }
1511
- return {
1512
- operation: { operationName },
1513
- warnings,
1514
- response: {
1515
- timestamp: currentDate,
1516
- modelId: this.modelId,
1517
- headers: responseHeaders
1518
- }
1519
- };
1520
- }
1521
- async doStatus(options) {
1522
- const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
1523
- const { operationName } = options.operation;
1524
- const { value: statusOperation, responseHeaders } = await postJsonToApi5({
1525
- url: `${this.config.baseURL}/models/${this.modelId}:fetchPredictOperation`,
1526
- headers: combineHeaders5(
1527
- await resolve5(this.config.headers),
1528
- options.headers
1529
- ),
1530
- body: {
1531
- operationName
1532
- },
1533
- successfulResponseHandler: createJsonResponseHandler5(
1534
- googleVertexOperationSchema
1535
- ),
1536
- failedResponseHandler: googleVertexFailedResponseHandler,
1537
- abortSignal: options.abortSignal,
1538
- fetch: this.config.fetch
1539
- });
1540
- if (!statusOperation.done) {
1541
- return {
1542
- status: "pending",
1543
- response: {
1544
- timestamp: currentDate,
1545
- modelId: this.modelId,
1546
- headers: responseHeaders
1547
- }
1548
- };
1549
- }
1550
- if (statusOperation.error) {
1551
- return {
1552
- status: "error",
1553
- error: `Video generation failed: ${statusOperation.error.message}`,
1554
- response: {
1555
- timestamp: currentDate,
1556
- modelId: this.modelId,
1557
- headers: responseHeaders
1558
- }
1559
- };
1560
- }
1561
- return this.buildCompletedResult({
1562
- finalOperation: statusOperation,
1563
- responseHeaders,
1564
- warnings: [],
1565
- currentDate
1566
- });
1567
- }
1030
+ get provider() {
1031
+ return this.config.provider;
1032
+ }
1033
+ get maxVideosPerCall() {
1034
+ return 4;
1035
+ }
1036
+ constructor(modelId, config) {
1037
+ this.modelId = modelId;
1038
+ this.config = config;
1039
+ this.specificationVersion = "v4";
1040
+ }
1041
+ async buildRequest(options) {
1042
+ const warnings = [];
1043
+ const googleVertexOptions = await parseProviderOptions({
1044
+ provider: "googleVertex",
1045
+ providerOptions: options.providerOptions,
1046
+ schema: googleVertexVideoModelOptionsSchema
1047
+ }) ?? await parseProviderOptions({
1048
+ provider: "vertex",
1049
+ providerOptions: options.providerOptions,
1050
+ schema: googleVertexVideoModelOptionsSchema
1051
+ });
1052
+ const instances = [{}];
1053
+ const instance = instances[0];
1054
+ if (options.prompt != null) instance.prompt = options.prompt;
1055
+ const startImage = resolveStartImage(options);
1056
+ if (startImage != null) {
1057
+ const image = convertFileToVertexImage(startImage, warnings);
1058
+ if (image != null) instance.image = image;
1059
+ }
1060
+ const lastFrameImage = getLastFrameImage(options);
1061
+ if (lastFrameImage != null) {
1062
+ const lastFrame = convertFileToVertexImage(lastFrameImage, warnings);
1063
+ if (lastFrame != null) instance.lastFrame = lastFrame;
1064
+ }
1065
+ const inputReferences = getInputReferences(options);
1066
+ if (inputReferences != null) instance.referenceImages = inputReferences.flatMap((reference) => {
1067
+ const converted = convertInputReferenceImage(reference, warnings);
1068
+ return converted != null ? [converted] : [];
1069
+ });
1070
+ else if (googleVertexOptions?.referenceImages != null) instance.referenceImages = googleVertexOptions.referenceImages;
1071
+ const parameters = { sampleCount: options.n };
1072
+ if (options.aspectRatio) parameters.aspectRatio = options.aspectRatio;
1073
+ if (options.resolution) parameters.resolution = {
1074
+ "1280x720": "720p",
1075
+ "1920x1080": "1080p",
1076
+ "3840x2160": "4k"
1077
+ }[options.resolution] || options.resolution;
1078
+ if (options.duration) parameters.durationSeconds = options.duration;
1079
+ if (options.seed) parameters.seed = options.seed;
1080
+ const generateAudio = options.generateAudio ?? googleVertexOptions?.generateAudio;
1081
+ if (generateAudio != null) parameters.generateAudio = generateAudio;
1082
+ if (googleVertexOptions != null) {
1083
+ const opts = googleVertexOptions;
1084
+ if (opts.personGeneration !== void 0 && opts.personGeneration !== null) parameters.personGeneration = opts.personGeneration;
1085
+ if (opts.negativePrompt !== void 0 && opts.negativePrompt !== null) parameters.negativePrompt = opts.negativePrompt;
1086
+ if (opts.gcsOutputDirectory !== void 0 && opts.gcsOutputDirectory !== null) parameters.gcsOutputDirectory = opts.gcsOutputDirectory;
1087
+ for (const [key, value] of Object.entries(opts)) if (![
1088
+ "pollIntervalMs",
1089
+ "pollTimeoutMs",
1090
+ "personGeneration",
1091
+ "negativePrompt",
1092
+ "generateAudio",
1093
+ "gcsOutputDirectory",
1094
+ "referenceImages"
1095
+ ].includes(key)) parameters[key] = value;
1096
+ }
1097
+ return {
1098
+ instances,
1099
+ parameters,
1100
+ warnings,
1101
+ googleVertexOptions
1102
+ };
1103
+ }
1104
+ buildCompletedResult({ finalOperation, responseHeaders, warnings, currentDate }) {
1105
+ const response = finalOperation.response;
1106
+ if (!response?.videos || response.videos.length === 0) throw new AISDKError({
1107
+ name: "VERTEX_VIDEO_GENERATION_ERROR",
1108
+ message: `No videos in response. Response: ${JSON.stringify(finalOperation)}`
1109
+ });
1110
+ const videos = [];
1111
+ const videoMetadata = [];
1112
+ for (const video of response.videos) if (video.bytesBase64Encoded) {
1113
+ videos.push({
1114
+ type: "base64",
1115
+ data: video.bytesBase64Encoded,
1116
+ mediaType: video.mimeType || "video/mp4"
1117
+ });
1118
+ videoMetadata.push({ mimeType: video.mimeType });
1119
+ } else if (video.gcsUri) {
1120
+ videos.push({
1121
+ type: "url",
1122
+ url: video.gcsUri,
1123
+ mediaType: video.mimeType || "video/mp4"
1124
+ });
1125
+ videoMetadata.push({
1126
+ gcsUri: video.gcsUri,
1127
+ mimeType: video.mimeType
1128
+ });
1129
+ }
1130
+ if (videos.length === 0) throw new AISDKError({
1131
+ name: "VERTEX_VIDEO_GENERATION_ERROR",
1132
+ message: "No valid videos in response"
1133
+ });
1134
+ return {
1135
+ status: "completed",
1136
+ videos,
1137
+ warnings,
1138
+ response: {
1139
+ timestamp: currentDate,
1140
+ modelId: this.modelId,
1141
+ headers: responseHeaders
1142
+ },
1143
+ providerMetadata: (() => {
1144
+ const payload = { videos: videoMetadata };
1145
+ return {
1146
+ googleVertex: payload,
1147
+ "google-vertex": payload,
1148
+ vertex: payload
1149
+ };
1150
+ })()
1151
+ };
1152
+ }
1153
+ async doStart(options) {
1154
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
1155
+ const { instances, parameters, warnings } = await this.buildRequest(options);
1156
+ const { value: operation, responseHeaders } = await postJsonToApi({
1157
+ url: `${this.config.baseURL}/models/${this.modelId}:predictLongRunning`,
1158
+ headers: combineHeaders(await resolve(this.config.headers), options.headers),
1159
+ body: {
1160
+ instances,
1161
+ parameters
1162
+ },
1163
+ successfulResponseHandler: createJsonResponseHandler(googleVertexOperationSchema),
1164
+ failedResponseHandler: googleVertexFailedResponseHandler,
1165
+ abortSignal: options.abortSignal,
1166
+ fetch: this.config.fetch
1167
+ });
1168
+ const operationName = operation.name;
1169
+ if (!operationName) throw new AISDKError({
1170
+ name: "VERTEX_VIDEO_GENERATION_ERROR",
1171
+ message: "No operation name returned from API"
1172
+ });
1173
+ return {
1174
+ operation: { operationName },
1175
+ warnings,
1176
+ response: {
1177
+ timestamp: currentDate,
1178
+ modelId: this.modelId,
1179
+ headers: responseHeaders
1180
+ }
1181
+ };
1182
+ }
1183
+ async doStatus(options) {
1184
+ const currentDate = this.config._internal?.currentDate?.() ?? /* @__PURE__ */ new Date();
1185
+ const { operationName } = options.operation;
1186
+ const { value: statusOperation, responseHeaders } = await postJsonToApi({
1187
+ url: `${this.config.baseURL}/models/${this.modelId}:fetchPredictOperation`,
1188
+ headers: combineHeaders(await resolve(this.config.headers), options.headers),
1189
+ body: { operationName },
1190
+ successfulResponseHandler: createJsonResponseHandler(googleVertexOperationSchema),
1191
+ failedResponseHandler: googleVertexFailedResponseHandler,
1192
+ abortSignal: options.abortSignal,
1193
+ fetch: this.config.fetch
1194
+ });
1195
+ if (!statusOperation.done) return {
1196
+ status: "pending",
1197
+ response: {
1198
+ timestamp: currentDate,
1199
+ modelId: this.modelId,
1200
+ headers: responseHeaders
1201
+ }
1202
+ };
1203
+ if (statusOperation.error) return {
1204
+ status: "error",
1205
+ error: `Video generation failed: ${statusOperation.error.message}`,
1206
+ response: {
1207
+ timestamp: currentDate,
1208
+ modelId: this.modelId,
1209
+ headers: responseHeaders
1210
+ }
1211
+ };
1212
+ return this.buildCompletedResult({
1213
+ finalOperation: statusOperation,
1214
+ responseHeaders,
1215
+ warnings: [],
1216
+ currentDate
1217
+ });
1218
+ }
1568
1219
  };
1569
- var googleVertexOperationSchema = z10.object({
1570
- name: z10.string().nullish(),
1571
- done: z10.boolean().nullish(),
1572
- error: z10.object({
1573
- code: z10.number().nullish(),
1574
- message: z10.string(),
1575
- status: z10.string().nullish()
1576
- }).nullish(),
1577
- response: z10.object({
1578
- videos: z10.array(
1579
- z10.object({
1580
- bytesBase64Encoded: z10.string().nullish(),
1581
- gcsUri: z10.string().nullish(),
1582
- mimeType: z10.string().nullish()
1583
- })
1584
- ).nullish(),
1585
- raiMediaFilteredCount: z10.number().nullish()
1586
- }).nullish()
1220
+ const googleVertexOperationSchema = z.object({
1221
+ name: z.string().nullish(),
1222
+ done: z.boolean().nullish(),
1223
+ error: z.object({
1224
+ code: z.number().nullish(),
1225
+ message: z.string(),
1226
+ status: z.string().nullish()
1227
+ }).nullish(),
1228
+ response: z.object({
1229
+ videos: z.array(z.object({
1230
+ bytesBase64Encoded: z.string().nullish(),
1231
+ gcsUri: z.string().nullish(),
1232
+ mimeType: z.string().nullish()
1233
+ })).nullish(),
1234
+ raiMediaFilteredCount: z.number().nullish()
1235
+ }).nullish()
1587
1236
  });
1588
-
1589
- // src/google-vertex-provider-base.ts
1590
- var EXPRESS_MODE_BASE_URL = "https://aiplatform.googleapis.com/v1/publishers/google";
1591
- var ENDPOINT_MODEL_PREFIX = "endpoints/";
1237
+ //#endregion
1238
+ //#region src/google-vertex-provider-base.ts
1239
+ const EXPRESS_MODE_BASE_URL = "https://aiplatform.googleapis.com/v1/publishers/google";
1240
+ const ENDPOINT_MODEL_PREFIX = "endpoints/";
1592
1241
  function isEndpointModelId(modelId) {
1593
- return modelId.startsWith(ENDPOINT_MODEL_PREFIX);
1242
+ return modelId.startsWith(ENDPOINT_MODEL_PREFIX);
1594
1243
  }
1595
1244
  function createExpressModeFetch(apiKey, customFetch) {
1596
- return async (url, init) => {
1597
- const modifiedInit = {
1598
- ...init,
1599
- headers: {
1600
- ...init?.headers ? normalizeHeaders(init.headers) : {},
1601
- "x-goog-api-key": apiKey
1602
- }
1603
- };
1604
- return (customFetch ?? fetch)(url.toString(), modifiedInit);
1605
- };
1245
+ return async (url, init) => {
1246
+ const modifiedInit = {
1247
+ ...init,
1248
+ headers: {
1249
+ ...init?.headers ? normalizeHeaders(init.headers) : {},
1250
+ "x-goog-api-key": apiKey
1251
+ }
1252
+ };
1253
+ return (customFetch ?? fetch)(url.toString(), modifiedInit);
1254
+ };
1606
1255
  }
1607
- function createGoogleVertex(options = {}) {
1608
- const apiKey = loadOptionalSetting({
1609
- settingValue: options.apiKey,
1610
- environmentVariableName: "GOOGLE_VERTEX_API_KEY"
1611
- });
1612
- const loadGoogleVertexProject = () => loadSetting({
1613
- settingValue: options.project,
1614
- settingName: "project",
1615
- environmentVariableName: "GOOGLE_VERTEX_PROJECT",
1616
- description: "Google Vertex project"
1617
- });
1618
- const loadGoogleVertexLocation = () => loadSetting({
1619
- settingValue: options.location,
1620
- settingName: "location",
1621
- environmentVariableName: "GOOGLE_VERTEX_LOCATION",
1622
- description: "Google Vertex location"
1623
- });
1624
- const loadBaseURL = ({ endpoint = false } = {}) => {
1625
- if (apiKey) {
1626
- return withoutTrailingSlash(options.baseURL) ?? EXPRESS_MODE_BASE_URL;
1627
- }
1628
- const region = loadGoogleVertexLocation();
1629
- const project = loadGoogleVertexProject();
1630
- const getHost = () => {
1631
- if (region === "global") {
1632
- return "aiplatform.googleapis.com";
1633
- } else if (region === "eu" || region === "us") {
1634
- return `aiplatform.${region}.rep.googleapis.com`;
1635
- } else {
1636
- return `${region}-aiplatform.googleapis.com`;
1637
- }
1638
- };
1639
- return withoutTrailingSlash(options.baseURL) ?? `https://${getHost()}/v1beta1/projects/${project}/locations/${region}${endpoint ? "" : "/publishers/google"}`;
1640
- };
1641
- const createConfig = (name, { endpoint = false } = {}) => {
1642
- const getHeaders = async () => {
1643
- const originalHeaders = await resolve6(options.headers ?? {});
1644
- return withUserAgentSuffix(
1645
- originalHeaders,
1646
- `ai-sdk-google-vertex/${VERSION}`
1647
- );
1648
- };
1649
- return {
1650
- provider: `google.vertex.${name}`,
1651
- headers: getHeaders,
1652
- fetch: apiKey ? createExpressModeFetch(apiKey, options.fetch) : options.fetch,
1653
- baseURL: loadBaseURL({ endpoint })
1654
- };
1655
- };
1656
- const createChatModel = (modelId) => {
1657
- const endpoint = isEndpointModelId(modelId);
1658
- if (endpoint && apiKey) {
1659
- throw new Error(
1660
- "Google Vertex tuned models do not support Express Mode API keys. Use standard Google Cloud credentials instead."
1661
- );
1662
- }
1663
- return new GoogleLanguageModel2(modelId, {
1664
- ...createConfig("chat", { endpoint }),
1665
- generateId: options.generateId ?? generateId,
1666
- supportedUrls: () => ({
1667
- "*": [
1668
- // HTTP URLs:
1669
- /^https?:\/\/.*$/,
1670
- // Google Cloud Storage URLs:
1671
- /^gs:\/\/.*$/
1672
- ]
1673
- }),
1674
- downloadToolResultFiles: {
1675
- maxBytes: options.toolResultDownloads?.maxBytes ?? 7 * 1024 * 1024,
1676
- supportsGoogleCloudStorageUrls: true
1677
- }
1678
- });
1679
- };
1680
- const createInteractionsModel = (modelIdOrAgent) => {
1681
- if (apiKey) {
1682
- throw new Error(
1683
- "Google Vertex Interactions models do not support Express Mode API keys. Use standard Google Cloud credentials instead."
1684
- );
1685
- }
1686
- return new GoogleInteractionsLanguageModel(
1687
- modelIdOrAgent,
1688
- {
1689
- // The Interactions API is a location-scoped resource
1690
- // (`.../locations/{region}/interactions`), so it uses the
1691
- // endpoint-style base URL without the `/publishers/google` suffix that
1692
- // the base-model paths carry.
1693
- ...createConfig("interactions", { endpoint: true }),
1694
- generateId: options.generateId ?? generateId
1695
- }
1696
- );
1697
- };
1698
- const createEmbeddingModel = (modelId) => new GoogleVertexEmbeddingModel(modelId, createConfig("embedding"));
1699
- const createImageModel = (modelId) => new GoogleVertexImageModel(modelId, {
1700
- ...createConfig("image"),
1701
- generateId: options.generateId ?? generateId
1702
- });
1703
- const createVideoModel = (modelId) => new GoogleVertexVideoModel(modelId, {
1704
- ...createConfig("video"),
1705
- generateId: options.generateId ?? generateId
1706
- });
1707
- const createSpeechModel = (modelId) => {
1708
- if (modelId.startsWith("chirp")) {
1709
- if (apiKey) {
1710
- throw new Error(
1711
- "Google Vertex Chirp speech models do not support Express Mode API keys. Use standard Google Cloud credentials instead."
1712
- );
1713
- }
1714
- const config = createConfig("speech");
1715
- return new GoogleVertexCloudTTSSpeechModel(modelId, {
1716
- provider: config.provider,
1717
- headers: config.headers,
1718
- fetch: config.fetch
1719
- });
1720
- }
1721
- return new GoogleSpeechModel(modelId, createConfig("speech"));
1722
- };
1723
- const createTranscriptionModel = (modelId) => {
1724
- if (apiKey) {
1725
- throw new Error(
1726
- "Google Vertex transcription models do not support Express Mode API keys. Use standard Google Cloud credentials instead."
1727
- );
1728
- }
1729
- const config = createConfig("transcription");
1730
- if (modelId.startsWith("gemini")) {
1731
- return new GoogleVertexGeminiTranscriptionModel(modelId, {
1732
- provider: config.provider,
1733
- baseURL: loadBaseURL(),
1734
- headers: config.headers,
1735
- fetch: config.fetch,
1736
- webSocket: options.webSocket,
1737
- project: loadGoogleVertexProject(),
1738
- location: loadGoogleVertexLocation()
1739
- });
1740
- }
1741
- return new GoogleVertexTranscriptionModel(modelId, {
1742
- provider: config.provider,
1743
- headers: config.headers,
1744
- fetch: config.fetch,
1745
- project: loadGoogleVertexProject(),
1746
- location: loadGoogleVertexLocation()
1747
- });
1748
- };
1749
- const provider = function(modelId) {
1750
- if (new.target) {
1751
- throw new Error(
1752
- "The Google Vertex AI model function cannot be called with the new keyword."
1753
- );
1754
- }
1755
- return createChatModel(modelId);
1756
- };
1757
- provider.specificationVersion = "v4";
1758
- provider.languageModel = createChatModel;
1759
- provider.interactions = createInteractionsModel;
1760
- provider.embeddingModel = createEmbeddingModel;
1761
- provider.textEmbeddingModel = createEmbeddingModel;
1762
- provider.image = createImageModel;
1763
- provider.imageModel = createImageModel;
1764
- provider.video = createVideoModel;
1765
- provider.videoModel = createVideoModel;
1766
- provider.speech = createSpeechModel;
1767
- provider.speechModel = createSpeechModel;
1768
- provider.transcription = createTranscriptionModel;
1769
- provider.transcriptionModel = createTranscriptionModel;
1770
- provider.tools = googleVertexTools;
1771
- return provider;
1256
+ /**
1257
+ * Create a Google Vertex AI provider instance.
1258
+ */
1259
+ function createGoogleVertex$1(options = {}) {
1260
+ const apiKey = loadOptionalSetting({
1261
+ settingValue: options.apiKey,
1262
+ environmentVariableName: "GOOGLE_VERTEX_API_KEY"
1263
+ });
1264
+ const loadGoogleVertexProject = () => loadSetting({
1265
+ settingValue: options.project,
1266
+ settingName: "project",
1267
+ environmentVariableName: "GOOGLE_VERTEX_PROJECT",
1268
+ description: "Google Vertex project"
1269
+ });
1270
+ const loadGoogleVertexLocation = () => {
1271
+ const location = loadSetting({
1272
+ settingValue: options.location,
1273
+ settingName: "location",
1274
+ environmentVariableName: "GOOGLE_VERTEX_LOCATION",
1275
+ description: "Google Vertex location"
1276
+ });
1277
+ if (!isValidHostnamePart(location)) throw new InvalidArgumentError({
1278
+ argument: "location",
1279
+ message: "Invalid Google Vertex location. Expected a single DNS label (letters, digits, and hyphens). Use `baseURL` for custom endpoints."
1280
+ });
1281
+ return location;
1282
+ };
1283
+ const loadBaseURL = ({ endpoint = false } = {}) => {
1284
+ if (apiKey) return withoutTrailingSlash(options.baseURL) ?? EXPRESS_MODE_BASE_URL;
1285
+ const baseURL = withoutTrailingSlash(options.baseURL);
1286
+ if (baseURL != null) return baseURL;
1287
+ const region = loadGoogleVertexLocation();
1288
+ const project = loadGoogleVertexProject();
1289
+ const getHost = () => {
1290
+ if (region === "global") return "aiplatform.googleapis.com";
1291
+ else if (region === "eu" || region === "us") return `aiplatform.${region}.rep.googleapis.com`;
1292
+ else return `${region}-aiplatform.googleapis.com`;
1293
+ };
1294
+ return `https://${getHost()}/v1beta1/projects/${project}/locations/${region}${endpoint ? "" : "/publishers/google"}`;
1295
+ };
1296
+ const createConfig = (name, { endpoint = false } = {}) => {
1297
+ const getHeaders = async () => {
1298
+ const originalHeaders = await resolve(options.headers ?? {});
1299
+ return withUserAgentSuffix(originalHeaders, `ai-sdk-google-vertex/${VERSION}`);
1300
+ };
1301
+ return {
1302
+ provider: `google.vertex.${name}`,
1303
+ headers: getHeaders,
1304
+ fetch: apiKey ? createExpressModeFetch(apiKey, options.fetch) : options.fetch,
1305
+ baseURL: loadBaseURL({ endpoint })
1306
+ };
1307
+ };
1308
+ const createChatModel = (modelId) => {
1309
+ const endpoint = isEndpointModelId(modelId);
1310
+ if (endpoint && apiKey) throw new Error("Google Vertex tuned models do not support Express Mode API keys. Use standard Google Cloud credentials instead.");
1311
+ return new GoogleLanguageModel(modelId, {
1312
+ ...createConfig("chat", { endpoint }),
1313
+ generateId: options.generateId ?? generateId,
1314
+ supportedUrls: () => ({ "*": [/^https?:\/\/.*$/, /^gs:\/\/.*$/] }),
1315
+ downloadToolResultFiles: {
1316
+ maxBytes: options.toolResultDownloads?.maxBytes ?? 7340032,
1317
+ supportsGoogleCloudStorageUrls: true
1318
+ }
1319
+ });
1320
+ };
1321
+ const createInteractionsModel = (modelIdOrAgent) => {
1322
+ if (apiKey) throw new Error("Google Vertex Interactions models do not support Express Mode API keys. Use standard Google Cloud credentials instead.");
1323
+ return new GoogleInteractionsLanguageModel(modelIdOrAgent, {
1324
+ ...createConfig("interactions", { endpoint: true }),
1325
+ generateId: options.generateId ?? generateId
1326
+ });
1327
+ };
1328
+ const createEmbeddingModel = (modelId) => new GoogleVertexEmbeddingModel(modelId, createConfig("embedding"));
1329
+ const createImageModel = (modelId) => new GoogleVertexImageModel(modelId, {
1330
+ ...createConfig("image"),
1331
+ generateId: options.generateId ?? generateId
1332
+ });
1333
+ const createVideoModel = (modelId) => new GoogleVertexVideoModel(modelId, {
1334
+ ...createConfig("video"),
1335
+ generateId: options.generateId ?? generateId
1336
+ });
1337
+ const createSpeechModel = (modelId) => {
1338
+ if (modelId.startsWith("chirp")) {
1339
+ if (apiKey) throw new Error("Google Vertex Chirp speech models do not support Express Mode API keys. Use standard Google Cloud credentials instead.");
1340
+ const config = createConfig("speech");
1341
+ return new GoogleVertexCloudTTSSpeechModel(modelId, {
1342
+ provider: config.provider,
1343
+ headers: config.headers,
1344
+ fetch: config.fetch
1345
+ });
1346
+ }
1347
+ return new GoogleSpeechModel(modelId, createConfig("speech"));
1348
+ };
1349
+ const createTranscriptionModel = (modelId) => {
1350
+ if (apiKey) throw new Error("Google Vertex transcription models do not support Express Mode API keys. Use standard Google Cloud credentials instead.");
1351
+ const config = createConfig("transcription");
1352
+ if (modelId.startsWith("gemini")) return new GoogleVertexGeminiTranscriptionModel(modelId, {
1353
+ provider: config.provider,
1354
+ baseURL: loadBaseURL(),
1355
+ headers: config.headers,
1356
+ fetch: config.fetch,
1357
+ webSocket: options.webSocket,
1358
+ project: loadGoogleVertexProject(),
1359
+ location: loadGoogleVertexLocation()
1360
+ });
1361
+ return new GoogleVertexTranscriptionModel(modelId, {
1362
+ provider: config.provider,
1363
+ headers: config.headers,
1364
+ fetch: config.fetch,
1365
+ project: loadGoogleVertexProject(),
1366
+ location: loadGoogleVertexLocation()
1367
+ });
1368
+ };
1369
+ const provider = function(modelId) {
1370
+ if (new.target) throw new Error("The Google Vertex AI model function cannot be called with the new keyword.");
1371
+ return createChatModel(modelId);
1372
+ };
1373
+ provider.specificationVersion = "v4";
1374
+ provider.languageModel = createChatModel;
1375
+ provider.interactions = createInteractionsModel;
1376
+ provider.embeddingModel = createEmbeddingModel;
1377
+ provider.textEmbeddingModel = createEmbeddingModel;
1378
+ provider.image = createImageModel;
1379
+ provider.imageModel = createImageModel;
1380
+ provider.video = createVideoModel;
1381
+ provider.videoModel = createVideoModel;
1382
+ provider.speech = createSpeechModel;
1383
+ provider.speechModel = createSpeechModel;
1384
+ provider.transcription = createTranscriptionModel;
1385
+ provider.transcriptionModel = createTranscriptionModel;
1386
+ provider.tools = googleVertexTools;
1387
+ return provider;
1772
1388
  }
1773
-
1774
- // src/edge/google-vertex-auth-edge.ts
1775
- import {
1776
- loadOptionalSetting as loadOptionalSetting2,
1777
- loadSetting as loadSetting2,
1778
- withUserAgentSuffix as withUserAgentSuffix2,
1779
- getRuntimeEnvironmentUserAgent
1780
- } from "@ai-sdk/provider-utils";
1781
- var loadCredentials = async () => {
1782
- try {
1783
- return {
1784
- clientEmail: loadSetting2({
1785
- settingValue: void 0,
1786
- settingName: "clientEmail",
1787
- environmentVariableName: "GOOGLE_CLIENT_EMAIL",
1788
- description: "Google client email"
1789
- }),
1790
- privateKey: loadSetting2({
1791
- settingValue: void 0,
1792
- settingName: "privateKey",
1793
- environmentVariableName: "GOOGLE_PRIVATE_KEY",
1794
- description: "Google private key"
1795
- }),
1796
- privateKeyId: loadOptionalSetting2({
1797
- settingValue: void 0,
1798
- environmentVariableName: "GOOGLE_PRIVATE_KEY_ID"
1799
- })
1800
- };
1801
- } catch (error) {
1802
- throw new Error(`Failed to load Google credentials: ${error.message}`);
1803
- }
1389
+ //#endregion
1390
+ //#region src/edge/google-vertex-auth-edge.ts
1391
+ const loadCredentials = async () => {
1392
+ try {
1393
+ return {
1394
+ clientEmail: loadSetting({
1395
+ settingValue: void 0,
1396
+ settingName: "clientEmail",
1397
+ environmentVariableName: "GOOGLE_CLIENT_EMAIL",
1398
+ description: "Google client email"
1399
+ }),
1400
+ privateKey: loadSetting({
1401
+ settingValue: void 0,
1402
+ settingName: "privateKey",
1403
+ environmentVariableName: "GOOGLE_PRIVATE_KEY",
1404
+ description: "Google private key"
1405
+ }),
1406
+ privateKeyId: loadOptionalSetting({
1407
+ settingValue: void 0,
1408
+ environmentVariableName: "GOOGLE_PRIVATE_KEY_ID"
1409
+ })
1410
+ };
1411
+ } catch (error) {
1412
+ throw new Error(`Failed to load Google credentials: ${error.message}`);
1413
+ }
1804
1414
  };
1805
- var base64url = (str) => {
1806
- return btoa(str).replace(/\+/g, "-").replace(/\//g, "_").replace(/=/g, "");
1415
+ const base64url = (str) => {
1416
+ return btoa(str).replace(/\+/g, "-").replace(/\//g, "_").replace(/=/g, "");
1807
1417
  };
1808
- var importPrivateKey = async (pemKey) => {
1809
- const pemHeader = "-----BEGIN PRIVATE KEY-----";
1810
- const pemFooter = "-----END PRIVATE KEY-----";
1811
- const pemContents = pemKey.replace(pemHeader, "").replace(pemFooter, "").replace(/\s/g, "");
1812
- const binaryString = atob(pemContents);
1813
- const binaryData = new Uint8Array(binaryString.length);
1814
- for (let i = 0; i < binaryString.length; i++) {
1815
- binaryData[i] = binaryString.charCodeAt(i);
1816
- }
1817
- return await crypto.subtle.importKey(
1818
- "pkcs8",
1819
- binaryData,
1820
- { name: "RSASSA-PKCS1-v1_5", hash: "SHA-256" },
1821
- true,
1822
- ["sign"]
1823
- );
1418
+ const importPrivateKey = async (pemKey) => {
1419
+ const pemContents = pemKey.replace("-----BEGIN PRIVATE KEY-----", "").replace("-----END PRIVATE KEY-----", "").replace(/\s/g, "");
1420
+ const binaryString = atob(pemContents);
1421
+ const binaryData = new Uint8Array(binaryString.length);
1422
+ for (let i = 0; i < binaryString.length; i++) binaryData[i] = binaryString.charCodeAt(i);
1423
+ return await crypto.subtle.importKey("pkcs8", binaryData, {
1424
+ name: "RSASSA-PKCS1-v1_5",
1425
+ hash: "SHA-256"
1426
+ }, true, ["sign"]);
1824
1427
  };
1825
- var buildJwt = async (credentials) => {
1826
- const now = Math.floor(Date.now() / 1e3);
1827
- const header = {
1828
- alg: "RS256",
1829
- typ: "JWT"
1830
- };
1831
- if (credentials.privateKeyId) {
1832
- header.kid = credentials.privateKeyId;
1833
- }
1834
- const payload = {
1835
- iss: credentials.clientEmail,
1836
- scope: "https://www.googleapis.com/auth/cloud-platform",
1837
- aud: "https://oauth2.googleapis.com/token",
1838
- exp: now + 3600,
1839
- iat: now
1840
- };
1841
- const privateKey = await importPrivateKey(credentials.privateKey);
1842
- const signingInput = `${base64url(JSON.stringify(header))}.${base64url(
1843
- JSON.stringify(payload)
1844
- )}`;
1845
- const encoder = new TextEncoder();
1846
- const data = encoder.encode(signingInput);
1847
- const signature = await crypto.subtle.sign(
1848
- "RSASSA-PKCS1-v1_5",
1849
- privateKey,
1850
- data
1851
- );
1852
- const signatureBase64 = base64url(
1853
- String.fromCharCode(...new Uint8Array(signature))
1854
- );
1855
- return `${base64url(JSON.stringify(header))}.${base64url(
1856
- JSON.stringify(payload)
1857
- )}.${signatureBase64}`;
1428
+ const buildJwt = async (credentials) => {
1429
+ const now = Math.floor(Date.now() / 1e3);
1430
+ const header = {
1431
+ alg: "RS256",
1432
+ typ: "JWT"
1433
+ };
1434
+ if (credentials.privateKeyId) header.kid = credentials.privateKeyId;
1435
+ const payload = {
1436
+ iss: credentials.clientEmail,
1437
+ scope: "https://www.googleapis.com/auth/cloud-platform",
1438
+ aud: "https://oauth2.googleapis.com/token",
1439
+ exp: now + 3600,
1440
+ iat: now
1441
+ };
1442
+ const privateKey = await importPrivateKey(credentials.privateKey);
1443
+ const signingInput = `${base64url(JSON.stringify(header))}.${base64url(JSON.stringify(payload))}`;
1444
+ const data = new TextEncoder().encode(signingInput);
1445
+ const signature = await crypto.subtle.sign("RSASSA-PKCS1-v1_5", privateKey, data);
1446
+ const signatureBase64 = base64url(String.fromCharCode(...new Uint8Array(signature)));
1447
+ return `${base64url(JSON.stringify(header))}.${base64url(JSON.stringify(payload))}.${signatureBase64}`;
1858
1448
  };
1449
+ /**
1450
+ * Generate an authentication token for Google Vertex AI in a manner compatible
1451
+ * with the Edge runtime.
1452
+ */
1859
1453
  async function generateAuthToken(credentials) {
1860
- const creds = credentials || await loadCredentials();
1861
- const jwt = await buildJwt(creds);
1862
- const response = await fetch("https://oauth2.googleapis.com/token", {
1863
- method: "POST",
1864
- headers: withUserAgentSuffix2(
1865
- { "Content-Type": "application/x-www-form-urlencoded" },
1866
- `ai-sdk-google-vertex/${VERSION}`,
1867
- getRuntimeEnvironmentUserAgent()
1868
- ),
1869
- body: new URLSearchParams({
1870
- grant_type: "urn:ietf:params:oauth:grant-type:jwt-bearer",
1871
- assertion: jwt
1872
- })
1873
- });
1874
- if (!response.ok) {
1875
- throw new Error(`Token request failed: ${response.statusText}`);
1876
- }
1877
- const data = await response.json();
1878
- return data.access_token;
1454
+ const creds = credentials || await loadCredentials();
1455
+ const jwt = await buildJwt(creds);
1456
+ const response = await fetch("https://oauth2.googleapis.com/token", {
1457
+ method: "POST",
1458
+ headers: withUserAgentSuffix({ "Content-Type": "application/x-www-form-urlencoded" }, `ai-sdk-google-vertex/${VERSION}`, getRuntimeEnvironmentUserAgent()),
1459
+ body: new URLSearchParams({
1460
+ grant_type: "urn:ietf:params:oauth:grant-type:jwt-bearer",
1461
+ assertion: jwt
1462
+ })
1463
+ });
1464
+ if (!response.ok) throw new Error(`Token request failed: ${response.statusText}`);
1465
+ return (await response.json()).access_token;
1879
1466
  }
1880
-
1881
- // src/edge/google-vertex-provider-edge.ts
1882
- function createGoogleVertex2(options = {}) {
1883
- const apiKey = loadOptionalSetting3({
1884
- settingValue: options.apiKey,
1885
- environmentVariableName: "GOOGLE_VERTEX_API_KEY"
1886
- });
1887
- if (apiKey) {
1888
- return createGoogleVertex(options);
1889
- }
1890
- return createGoogleVertex({
1891
- ...options,
1892
- headers: async () => ({
1893
- Authorization: `Bearer ${await generateAuthToken(
1894
- options.googleCredentials
1895
- )}`,
1896
- ...await resolve7(options.headers)
1897
- })
1898
- });
1467
+ //#endregion
1468
+ //#region src/edge/google-vertex-provider-edge.ts
1469
+ function createGoogleVertex(options = {}) {
1470
+ if (loadOptionalSetting({
1471
+ settingValue: options.apiKey,
1472
+ environmentVariableName: "GOOGLE_VERTEX_API_KEY"
1473
+ })) return createGoogleVertex$1(options);
1474
+ return createGoogleVertex$1({
1475
+ ...options,
1476
+ headers: async () => ({
1477
+ Authorization: `Bearer ${await generateAuthToken(options.googleCredentials)}`,
1478
+ ...await resolve(options.headers)
1479
+ })
1480
+ });
1899
1481
  }
1900
- var googleVertex = createGoogleVertex2();
1901
- export {
1902
- createGoogleVertex2 as createGoogleVertex,
1903
- createGoogleVertex2 as createVertex,
1904
- googleVertex,
1905
- googleVertex as vertex
1906
- };
1482
+ /**
1483
+ * Default Google Vertex AI provider instance.
1484
+ */
1485
+ const googleVertex = createGoogleVertex();
1486
+ //#endregion
1487
+ export { createGoogleVertex, createGoogleVertex as createVertex, googleVertex, googleVertex as vertex };
1488
+
1907
1489
  //# sourceMappingURL=index.js.map