@ai-sdk/google 3.0.114 → 3.0.118
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/dist/index.d.mts +64 -9
- package/dist/index.d.ts +64 -9
- package/dist/index.js +1011 -719
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +948 -648
- package/dist/index.mjs.map +1 -1
- package/dist/internal/index.d.mts +7 -6
- package/dist/internal/index.d.ts +7 -6
- package/dist/internal/index.js +150 -52
- package/dist/internal/index.js.map +1 -1
- package/dist/internal/index.mjs +150 -52
- package/dist/internal/index.mjs.map +1 -1
- package/package.json +2 -2
- package/src/convert-json-schema-to-openapi-schema.ts +123 -7
- package/src/google-generative-ai-language-model.ts +99 -39
- package/src/google-provider.ts +24 -0
- package/src/index.ts +5 -0
- package/src/transcription/google-transcription-model-options.ts +51 -0
- package/src/transcription/google-transcription-model.ts +243 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ai-sdk/google",
|
|
3
|
-
"version": "3.0.
|
|
3
|
+
"version": "3.0.118",
|
|
4
4
|
"license": "Apache-2.0",
|
|
5
5
|
"sideEffects": false,
|
|
6
6
|
"main": "./dist/index.js",
|
|
@@ -37,7 +37,7 @@
|
|
|
37
37
|
},
|
|
38
38
|
"dependencies": {
|
|
39
39
|
"@ai-sdk/provider": "3.0.15",
|
|
40
|
-
"@ai-sdk/provider-utils": "4.0.
|
|
40
|
+
"@ai-sdk/provider-utils": "4.0.49"
|
|
41
41
|
},
|
|
42
42
|
"devDependencies": {
|
|
43
43
|
"@types/node": "20.17.24",
|
|
@@ -100,10 +100,6 @@ function convertJSONSchemaDefinition(
|
|
|
100
100
|
if (required) result.required = required;
|
|
101
101
|
if (format) result.format = format;
|
|
102
102
|
|
|
103
|
-
if (constValue !== undefined) {
|
|
104
|
-
result.enum = [constValue];
|
|
105
|
-
}
|
|
106
|
-
|
|
107
103
|
// Handle type
|
|
108
104
|
if (type) {
|
|
109
105
|
if (Array.isArray(type)) {
|
|
@@ -125,9 +121,11 @@ function convertJSONSchemaDefinition(
|
|
|
125
121
|
}
|
|
126
122
|
}
|
|
127
123
|
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
124
|
+
const values =
|
|
125
|
+
enumValues ?? (constValue !== undefined ? [constValue] : undefined);
|
|
126
|
+
|
|
127
|
+
if (values !== undefined) {
|
|
128
|
+
addEnumToSchema({ values, type, result });
|
|
131
129
|
}
|
|
132
130
|
|
|
133
131
|
if (properties != null) {
|
|
@@ -201,6 +199,123 @@ function convertJSONSchemaDefinition(
|
|
|
201
199
|
return result;
|
|
202
200
|
}
|
|
203
201
|
|
|
202
|
+
type EnumValues = NonNullable<JSONSchema7['enum']>;
|
|
203
|
+
type EnumType = 'string' | 'number' | 'integer' | 'boolean';
|
|
204
|
+
type GoogleEnumSchema = {
|
|
205
|
+
type?: JSONSchema7['type'];
|
|
206
|
+
enum?: JSONSchema7['enum'];
|
|
207
|
+
format?: JSONSchema7['format'];
|
|
208
|
+
anyOf?: JSONSchema7['anyOf'];
|
|
209
|
+
nullable?: boolean;
|
|
210
|
+
};
|
|
211
|
+
|
|
212
|
+
function addEnumToSchema({
|
|
213
|
+
values,
|
|
214
|
+
type,
|
|
215
|
+
result,
|
|
216
|
+
}: {
|
|
217
|
+
values: EnumValues;
|
|
218
|
+
type: JSONSchema7['type'];
|
|
219
|
+
result: GoogleEnumSchema;
|
|
220
|
+
}) {
|
|
221
|
+
const nullable =
|
|
222
|
+
(Array.isArray(type) && type.includes('null')) ||
|
|
223
|
+
(type === undefined && values.includes(null));
|
|
224
|
+
|
|
225
|
+
// Gemini uses nullable instead of a null enum member.
|
|
226
|
+
const enumValues = nullable ? values.filter(value => value !== null) : values;
|
|
227
|
+
|
|
228
|
+
if (values.length > 0 && values.every(value => value === null)) {
|
|
229
|
+
const typeAllowsNull =
|
|
230
|
+
type === undefined ||
|
|
231
|
+
type === 'null' ||
|
|
232
|
+
(Array.isArray(type) && type.includes('null'));
|
|
233
|
+
|
|
234
|
+
if (typeAllowsNull) {
|
|
235
|
+
result.type = 'null';
|
|
236
|
+
if (Array.isArray(type)) {
|
|
237
|
+
delete result.anyOf;
|
|
238
|
+
}
|
|
239
|
+
return;
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
const enumType = getEnumType({ values: enumValues, type });
|
|
244
|
+
|
|
245
|
+
if (enumType === undefined) {
|
|
246
|
+
throw new UnsupportedFunctionalityError({
|
|
247
|
+
functionality: 'JSON Schema enum with mixed or unsupported values',
|
|
248
|
+
message:
|
|
249
|
+
'Google does not support this JSON Schema enum. Enum values must share one supported primitive type and match the schema type.',
|
|
250
|
+
});
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
result.type = enumType;
|
|
254
|
+
|
|
255
|
+
// The earlier type-array conversion created anyOf. The enum gives us one
|
|
256
|
+
// concrete value type, so store that type directly.
|
|
257
|
+
if (Array.isArray(type)) {
|
|
258
|
+
delete result.anyOf;
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
if (nullable) {
|
|
262
|
+
result.nullable = true;
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
if (enumType === 'string') {
|
|
266
|
+
result.enum = enumValues;
|
|
267
|
+
} else {
|
|
268
|
+
result.format = 'enum';
|
|
269
|
+
result.enum = enumValues.map(String);
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
function getEnumType({
|
|
274
|
+
values,
|
|
275
|
+
type,
|
|
276
|
+
}: {
|
|
277
|
+
values: EnumValues;
|
|
278
|
+
type: JSONSchema7['type'];
|
|
279
|
+
}): EnumType | undefined {
|
|
280
|
+
if (values.length === 0) {
|
|
281
|
+
return undefined;
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
const typeAllows = (enumType: EnumType) =>
|
|
285
|
+
type === undefined ||
|
|
286
|
+
type === enumType ||
|
|
287
|
+
(Array.isArray(type) && type.includes(enumType));
|
|
288
|
+
|
|
289
|
+
if (
|
|
290
|
+
typeAllows('string') &&
|
|
291
|
+
values.every(value => typeof value === 'string')
|
|
292
|
+
) {
|
|
293
|
+
return 'string';
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
if (
|
|
297
|
+
(typeAllows('number') || typeAllows('integer')) &&
|
|
298
|
+
values.every(value => typeof value === 'number' && Number.isFinite(value))
|
|
299
|
+
) {
|
|
300
|
+
if (typeAllows('number')) {
|
|
301
|
+
return 'number';
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
if (values.every(value => Number.isInteger(value))) {
|
|
305
|
+
return 'integer';
|
|
306
|
+
}
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
if (
|
|
310
|
+
typeAllows('boolean') &&
|
|
311
|
+
values.every(value => typeof value === 'boolean')
|
|
312
|
+
) {
|
|
313
|
+
return 'boolean';
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
return undefined;
|
|
317
|
+
}
|
|
318
|
+
|
|
204
319
|
function convertJSONSchemaReference({
|
|
205
320
|
jsonSchema,
|
|
206
321
|
reference,
|
|
@@ -318,6 +433,7 @@ function throwUnsupportedReference(reference: string): never {
|
|
|
318
433
|
'Google schema conversion only supports references to direct children of root-level $defs or definitions.',
|
|
319
434
|
});
|
|
320
435
|
}
|
|
436
|
+
|
|
321
437
|
function isEmptyObjectSchema(jsonSchema: JSONSchema7Definition): boolean {
|
|
322
438
|
return (
|
|
323
439
|
jsonSchema != null &&
|
|
@@ -58,6 +58,8 @@ const configurableSafetySettingCategories = [
|
|
|
58
58
|
'HARM_CATEGORY_SEXUALLY_EXPLICIT',
|
|
59
59
|
] as const;
|
|
60
60
|
|
|
61
|
+
const gemini25ModelPattern = /(^|\/)gemini-2\.5(?:[.-]|$)/i;
|
|
62
|
+
|
|
61
63
|
type GoogleGenerativeAIConfig = {
|
|
62
64
|
provider: string;
|
|
63
65
|
baseURL: string;
|
|
@@ -230,6 +232,22 @@ export class GoogleGenerativeAILanguageModel implements LanguageModelV3 {
|
|
|
230
232
|
}
|
|
231
233
|
|
|
232
234
|
const isGemmaModel = this.modelId.toLowerCase().startsWith('gemma-');
|
|
235
|
+
const isGemini25DeveloperApiModel =
|
|
236
|
+
!isVertexProvider && gemini25ModelPattern.test(this.modelId);
|
|
237
|
+
|
|
238
|
+
if (isGemini25DeveloperApiModel && frequencyPenalty != null) {
|
|
239
|
+
warnings.push({
|
|
240
|
+
type: 'unsupported',
|
|
241
|
+
feature: 'frequencyPenalty',
|
|
242
|
+
});
|
|
243
|
+
}
|
|
244
|
+
if (isGemini25DeveloperApiModel && presencePenalty != null) {
|
|
245
|
+
warnings.push({
|
|
246
|
+
type: 'unsupported',
|
|
247
|
+
feature: 'presencePenalty',
|
|
248
|
+
});
|
|
249
|
+
}
|
|
250
|
+
|
|
233
251
|
const { usesGemini3Features } = getGoogleModelCapabilities(this.modelId);
|
|
234
252
|
|
|
235
253
|
const { contents, systemInstruction } = convertToGoogleGenerativeAIMessages(
|
|
@@ -296,8 +314,12 @@ export class GoogleGenerativeAILanguageModel implements LanguageModelV3 {
|
|
|
296
314
|
temperature,
|
|
297
315
|
topK,
|
|
298
316
|
topP,
|
|
299
|
-
frequencyPenalty
|
|
300
|
-
|
|
317
|
+
frequencyPenalty: isGemini25DeveloperApiModel
|
|
318
|
+
? undefined
|
|
319
|
+
: frequencyPenalty,
|
|
320
|
+
presencePenalty: isGemini25DeveloperApiModel
|
|
321
|
+
? undefined
|
|
322
|
+
: presencePenalty,
|
|
301
323
|
stopSequences,
|
|
302
324
|
seed,
|
|
303
325
|
|
|
@@ -368,11 +390,16 @@ export class GoogleGenerativeAILanguageModel implements LanguageModelV3 {
|
|
|
368
390
|
fetch: this.config.fetch,
|
|
369
391
|
});
|
|
370
392
|
|
|
371
|
-
const candidate = response.candidates[0];
|
|
393
|
+
const candidate = response.candidates?.[0];
|
|
394
|
+
const promptBlockReason = response.promptFeedback?.blockReason;
|
|
395
|
+
const isPromptBlocked =
|
|
396
|
+
candidate?.finishReason == null && promptBlockReason != null;
|
|
397
|
+
const rawFinishReason =
|
|
398
|
+
candidate?.finishReason ?? promptBlockReason ?? undefined;
|
|
372
399
|
const content: Array<LanguageModelV3Content> = [];
|
|
373
400
|
|
|
374
401
|
// map ordered parts to content:
|
|
375
|
-
const parts = candidate
|
|
402
|
+
const parts = candidate?.content?.parts ?? [];
|
|
376
403
|
|
|
377
404
|
const usageMetadata = response.usageMetadata;
|
|
378
405
|
|
|
@@ -515,7 +542,7 @@ export class GoogleGenerativeAILanguageModel implements LanguageModelV3 {
|
|
|
515
542
|
|
|
516
543
|
const sources =
|
|
517
544
|
extractSources({
|
|
518
|
-
groundingMetadata: candidate
|
|
545
|
+
groundingMetadata: candidate?.groundingMetadata,
|
|
519
546
|
generateId: this.config.generateId,
|
|
520
547
|
}) ?? [];
|
|
521
548
|
for (const source of sources) {
|
|
@@ -525,25 +552,27 @@ export class GoogleGenerativeAILanguageModel implements LanguageModelV3 {
|
|
|
525
552
|
return {
|
|
526
553
|
content,
|
|
527
554
|
finishReason: {
|
|
528
|
-
unified:
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
555
|
+
unified: isPromptBlocked
|
|
556
|
+
? 'content-filter'
|
|
557
|
+
: mapGoogleGenerativeAIFinishReason({
|
|
558
|
+
finishReason: rawFinishReason,
|
|
559
|
+
// Only count client-executed tool calls for finish reason determination.
|
|
560
|
+
hasToolCalls: content.some(
|
|
561
|
+
part => part.type === 'tool-call' && !part.providerExecuted,
|
|
562
|
+
),
|
|
563
|
+
}),
|
|
564
|
+
raw: rawFinishReason,
|
|
536
565
|
},
|
|
537
566
|
usage: convertGoogleGenerativeAIUsage(usageMetadata),
|
|
538
567
|
warnings,
|
|
539
568
|
providerMetadata: {
|
|
540
569
|
[providerOptionsName]: {
|
|
541
570
|
promptFeedback: response.promptFeedback ?? null,
|
|
542
|
-
groundingMetadata: candidate
|
|
543
|
-
urlContextMetadata: candidate
|
|
544
|
-
safetyRatings: candidate
|
|
571
|
+
groundingMetadata: candidate?.groundingMetadata ?? null,
|
|
572
|
+
urlContextMetadata: candidate?.urlContextMetadata ?? null,
|
|
573
|
+
safetyRatings: candidate?.safetyRatings ?? null,
|
|
545
574
|
usageMetadata: usageMetadata ?? null,
|
|
546
|
-
finishMessage: candidate
|
|
575
|
+
finishMessage: candidate?.finishMessage ?? null,
|
|
547
576
|
serviceTier: usageMetadata?.serviceTier ?? null,
|
|
548
577
|
} satisfies GoogleGenerativeAIProviderMetadata,
|
|
549
578
|
},
|
|
@@ -689,6 +718,24 @@ export class GoogleGenerativeAILanguageModel implements LanguageModelV3 {
|
|
|
689
718
|
|
|
690
719
|
// sometimes the API returns an empty candidates array
|
|
691
720
|
if (candidate == null) {
|
|
721
|
+
const promptBlockReason = value.promptFeedback?.blockReason;
|
|
722
|
+
if (promptBlockReason != null) {
|
|
723
|
+
finishReason = {
|
|
724
|
+
unified: 'content-filter',
|
|
725
|
+
raw: promptBlockReason,
|
|
726
|
+
};
|
|
727
|
+
providerMetadata = {
|
|
728
|
+
[providerOptionsName]: {
|
|
729
|
+
promptFeedback: value.promptFeedback ?? null,
|
|
730
|
+
groundingMetadata: lastGroundingMetadata,
|
|
731
|
+
urlContextMetadata: lastUrlContextMetadata,
|
|
732
|
+
safetyRatings: null,
|
|
733
|
+
usageMetadata: usageMetadata ?? null,
|
|
734
|
+
finishMessage: null,
|
|
735
|
+
serviceTier: usage?.serviceTier ?? null,
|
|
736
|
+
} satisfies GoogleGenerativeAIProviderMetadata,
|
|
737
|
+
};
|
|
738
|
+
}
|
|
692
739
|
return;
|
|
693
740
|
}
|
|
694
741
|
|
|
@@ -1078,13 +1125,21 @@ export class GoogleGenerativeAILanguageModel implements LanguageModelV3 {
|
|
|
1078
1125
|
}
|
|
1079
1126
|
}
|
|
1080
1127
|
|
|
1081
|
-
|
|
1128
|
+
const promptBlockReason = value.promptFeedback?.blockReason;
|
|
1129
|
+
const isPromptBlocked =
|
|
1130
|
+
candidate.finishReason == null && promptBlockReason != null;
|
|
1131
|
+
const rawFinishReason =
|
|
1132
|
+
candidate.finishReason ?? promptBlockReason ?? undefined;
|
|
1133
|
+
|
|
1134
|
+
if (rawFinishReason != null) {
|
|
1082
1135
|
finishReason = {
|
|
1083
|
-
unified:
|
|
1084
|
-
|
|
1085
|
-
|
|
1086
|
-
|
|
1087
|
-
|
|
1136
|
+
unified: isPromptBlocked
|
|
1137
|
+
? 'content-filter'
|
|
1138
|
+
: mapGoogleGenerativeAIFinishReason({
|
|
1139
|
+
finishReason: rawFinishReason,
|
|
1140
|
+
hasToolCalls,
|
|
1141
|
+
}),
|
|
1142
|
+
raw: rawFinishReason,
|
|
1088
1143
|
};
|
|
1089
1144
|
|
|
1090
1145
|
providerMetadata = {
|
|
@@ -1464,16 +1519,18 @@ const responseSchema = lazySchema(() =>
|
|
|
1464
1519
|
zodSchema(
|
|
1465
1520
|
z.object({
|
|
1466
1521
|
responseId: z.string().nullish(),
|
|
1467
|
-
candidates: z
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1472
|
-
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1522
|
+
candidates: z
|
|
1523
|
+
.array(
|
|
1524
|
+
z.object({
|
|
1525
|
+
content: getContentSchema().nullish().or(z.object({}).strict()),
|
|
1526
|
+
finishReason: z.string().nullish(),
|
|
1527
|
+
finishMessage: z.string().nullish(),
|
|
1528
|
+
safetyRatings: z.array(getSafetyRatingSchema()).nullish(),
|
|
1529
|
+
groundingMetadata: getGroundingMetadataSchema().nullish(),
|
|
1530
|
+
urlContextMetadata: getUrlContextMetadataSchema().nullish(),
|
|
1531
|
+
}),
|
|
1532
|
+
)
|
|
1533
|
+
.nullish(),
|
|
1477
1534
|
usageMetadata: usageSchema.nullish(),
|
|
1478
1535
|
promptFeedback: z
|
|
1479
1536
|
.object({
|
|
@@ -1485,11 +1542,14 @@ const responseSchema = lazySchema(() =>
|
|
|
1485
1542
|
),
|
|
1486
1543
|
);
|
|
1487
1544
|
|
|
1488
|
-
type
|
|
1489
|
-
InferSchema<typeof responseSchema>['candidates']
|
|
1490
|
-
|
|
1545
|
+
type CandidateSchema = NonNullable<
|
|
1546
|
+
InferSchema<typeof responseSchema>['candidates']
|
|
1547
|
+
>[number];
|
|
1548
|
+
|
|
1549
|
+
type ContentSchema = NonNullable<CandidateSchema['content']>;
|
|
1550
|
+
|
|
1491
1551
|
export type GroundingMetadataSchema = NonNullable<
|
|
1492
|
-
|
|
1552
|
+
CandidateSchema['groundingMetadata']
|
|
1493
1553
|
>;
|
|
1494
1554
|
|
|
1495
1555
|
type GroundingChunkSchema = NonNullable<
|
|
@@ -1497,11 +1557,11 @@ type GroundingChunkSchema = NonNullable<
|
|
|
1497
1557
|
>[number];
|
|
1498
1558
|
|
|
1499
1559
|
export type UrlContextMetadataSchema = NonNullable<
|
|
1500
|
-
|
|
1560
|
+
CandidateSchema['urlContextMetadata']
|
|
1501
1561
|
>;
|
|
1502
1562
|
|
|
1503
1563
|
export type SafetyRatingSchema = NonNullable<
|
|
1504
|
-
|
|
1564
|
+
CandidateSchema['safetyRatings']
|
|
1505
1565
|
>[number];
|
|
1506
1566
|
|
|
1507
1567
|
export type PromptFeedbackSchema = NonNullable<
|
package/src/google-provider.ts
CHANGED
|
@@ -4,6 +4,7 @@ import type {
|
|
|
4
4
|
ImageModelV3,
|
|
5
5
|
LanguageModelV3,
|
|
6
6
|
ProviderV3,
|
|
7
|
+
TranscriptionModelV3,
|
|
7
8
|
} from '@ai-sdk/provider';
|
|
8
9
|
import {
|
|
9
10
|
generateId,
|
|
@@ -32,6 +33,8 @@ import {
|
|
|
32
33
|
} from './interactions/google-interactions-language-model';
|
|
33
34
|
import type { GoogleInteractionsModelId } from './interactions/google-interactions-language-model-options';
|
|
34
35
|
import type { GoogleInteractionsAgentName } from './interactions/google-interactions-agent';
|
|
36
|
+
import { GoogleTranscriptionModel } from './transcription/google-transcription-model';
|
|
37
|
+
import type { GoogleTranscriptionModelId } from './transcription/google-transcription-model-options';
|
|
35
38
|
|
|
36
39
|
export interface GoogleGenerativeAIProvider extends ProviderV3 {
|
|
37
40
|
(modelId: GoogleGenerativeAIModelId): LanguageModelV3;
|
|
@@ -75,6 +78,17 @@ export interface GoogleGenerativeAIProvider extends ProviderV3 {
|
|
|
75
78
|
modelId: GoogleGenerativeAIEmbeddingModelId,
|
|
76
79
|
): EmbeddingModelV3;
|
|
77
80
|
|
|
81
|
+
/**
|
|
82
|
+
* Creates a model for transcription (speech-to-text), e.g.
|
|
83
|
+
* `gemini-3.5-transcribe`.
|
|
84
|
+
*/
|
|
85
|
+
transcription(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Creates a model for transcription (speech-to-text).
|
|
89
|
+
*/
|
|
90
|
+
transcriptionModel(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
91
|
+
|
|
78
92
|
/**
|
|
79
93
|
* Creates a model for video generation.
|
|
80
94
|
*/
|
|
@@ -245,6 +259,14 @@ export function createGoogleGenerativeAI(
|
|
|
245
259
|
fetch: options.fetch,
|
|
246
260
|
});
|
|
247
261
|
|
|
262
|
+
const createTranscriptionModel = (modelId: GoogleTranscriptionModelId) =>
|
|
263
|
+
new GoogleTranscriptionModel(modelId, {
|
|
264
|
+
provider: `${providerName}.transcription`,
|
|
265
|
+
baseURL,
|
|
266
|
+
headers: getHeaders,
|
|
267
|
+
fetch: options.fetch,
|
|
268
|
+
});
|
|
269
|
+
|
|
248
270
|
const createVideoModel = (modelId: GoogleGenerativeAIVideoModelId) =>
|
|
249
271
|
new GoogleGenerativeAIVideoModel(modelId, {
|
|
250
272
|
provider: providerName,
|
|
@@ -291,6 +313,8 @@ export function createGoogleGenerativeAI(
|
|
|
291
313
|
provider.textEmbeddingModel = createEmbeddingModel;
|
|
292
314
|
provider.image = createImageModel;
|
|
293
315
|
provider.imageModel = createImageModel;
|
|
316
|
+
provider.transcription = createTranscriptionModel;
|
|
317
|
+
provider.transcriptionModel = createTranscriptionModel;
|
|
294
318
|
provider.video = createVideoModel;
|
|
295
319
|
provider.videoModel = createVideoModel;
|
|
296
320
|
provider.interactions = createInteractionsModel;
|
package/src/index.ts
CHANGED
|
@@ -27,6 +27,11 @@ export type {
|
|
|
27
27
|
} from './interactions/google-interactions-language-model-options';
|
|
28
28
|
export type { GoogleInteractionsProviderMetadata } from './interactions/google-interactions-provider-metadata';
|
|
29
29
|
export type { GoogleInteractionsAgentName } from './interactions/google-interactions-agent';
|
|
30
|
+
export { GoogleTranscriptionModel } from './transcription/google-transcription-model';
|
|
31
|
+
export type {
|
|
32
|
+
GoogleTranscriptionModelId,
|
|
33
|
+
GoogleTranscriptionModelOptions,
|
|
34
|
+
} from './transcription/google-transcription-model-options';
|
|
30
35
|
export { createGoogleGenerativeAI, google } from './google-provider';
|
|
31
36
|
export type {
|
|
32
37
|
GoogleGenerativeAIProvider,
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { z } from 'zod/v4';
|
|
2
|
+
|
|
3
|
+
export type GoogleTranscriptionModelId =
|
|
4
|
+
| 'gemini-3.5-transcribe'
|
|
5
|
+
| 'gemini-3.5-transcribe-live'
|
|
6
|
+
| (string & {});
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Speech recognition options for Gemini transcription models
|
|
10
|
+
* (`gemini-3.5-transcribe`). Maps onto Google's `AudioTranscriptionConfig`.
|
|
11
|
+
* The live variant (`gemini-3.5-transcribe-live`) requires streaming
|
|
12
|
+
* transcription, which is only available in AI SDK v7.
|
|
13
|
+
*/
|
|
14
|
+
export const googleTranscriptionModelOptions = z.object({
|
|
15
|
+
/**
|
|
16
|
+
* BCP-47 language codes providing hints about the languages present in the
|
|
17
|
+
* audio. If omitted or empty, defaults to automatic language detection.
|
|
18
|
+
*/
|
|
19
|
+
languageCodes: z.array(z.string()).optional(),
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Custom vocabulary phrases, which bias the speech recognition model
|
|
23
|
+
* toward recognizing specific terms.
|
|
24
|
+
*/
|
|
25
|
+
customVocabulary: z.array(z.string()).optional(),
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Enables word-level timestamp generation.
|
|
29
|
+
*/
|
|
30
|
+
wordTimestamp: z.boolean().optional(),
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Enables speaker diarization.
|
|
34
|
+
*/
|
|
35
|
+
diarization: z.boolean().optional(),
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Transcription output formatting mode.
|
|
39
|
+
*
|
|
40
|
+
* - `VERBATIM` (default): exact literal transcript preserving filler
|
|
41
|
+
* words, repetitions, and false starts.
|
|
42
|
+
* - `SMART`: cleans up and structures the transcript in real time —
|
|
43
|
+
* disfluency removal, inline self-corrections, structured formatting
|
|
44
|
+
* (lists, numbers, dates, paragraph breaks), and grammar/casing polish.
|
|
45
|
+
*/
|
|
46
|
+
mode: z.enum(['SMART', 'VERBATIM']).optional(),
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
export type GoogleTranscriptionModelOptions = z.infer<
|
|
50
|
+
typeof googleTranscriptionModelOptions
|
|
51
|
+
>;
|