@ai-sdk/google 3.0.114 → 3.0.118
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/dist/index.d.mts +64 -9
- package/dist/index.d.ts +64 -9
- package/dist/index.js +1011 -719
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +948 -648
- package/dist/index.mjs.map +1 -1
- package/dist/internal/index.d.mts +7 -6
- package/dist/internal/index.d.ts +7 -6
- package/dist/internal/index.js +150 -52
- package/dist/internal/index.js.map +1 -1
- package/dist/internal/index.mjs +150 -52
- package/dist/internal/index.mjs.map +1 -1
- package/package.json +2 -2
- package/src/convert-json-schema-to-openapi-schema.ts +123 -7
- package/src/google-generative-ai-language-model.ts +99 -39
- package/src/google-provider.ts +24 -0
- package/src/index.ts +5 -0
- package/src/transcription/google-transcription-model-options.ts +51 -0
- package/src/transcription/google-transcription-model.ts +243 -0
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,31 @@
|
|
|
1
1
|
# @ai-sdk/google
|
|
2
2
|
|
|
3
|
+
## 3.0.118
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- 35430d7: Omit unsupported frequency and presence penalties from Gemini 2.5 requests and return warnings instead.
|
|
8
|
+
|
|
9
|
+
## 3.0.117
|
|
10
|
+
|
|
11
|
+
### Patch Changes
|
|
12
|
+
|
|
13
|
+
- b068651: Surface prompt-level Google safety blocks without candidates as content-filter results with prompt feedback metadata.
|
|
14
|
+
|
|
15
|
+
## 3.0.116
|
|
16
|
+
|
|
17
|
+
### Patch Changes
|
|
18
|
+
|
|
19
|
+
- 9a1656e: fix(google): convert enum values to the Gemini schema format
|
|
20
|
+
- cfdc8df: feat (provider/google, provider/google-vertex): Gemini 3.5 Transcribe support — unary transcription (`gemini-3.5-transcribe`) via generateContent with language detection, speaker diarization, word timestamps, custom vocabulary, and `mode: 'VERBATIM' | 'SMART'` transcription formatting
|
|
21
|
+
|
|
22
|
+
## 3.0.115
|
|
23
|
+
|
|
24
|
+
### Patch Changes
|
|
25
|
+
|
|
26
|
+
- Updated dependencies [9a521b9]
|
|
27
|
+
- @ai-sdk/provider-utils@4.0.49
|
|
28
|
+
|
|
3
29
|
## 3.0.114
|
|
4
30
|
|
|
5
31
|
### Patch Changes
|
package/dist/index.d.mts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import * as _ai_sdk_provider_utils from '@ai-sdk/provider-utils';
|
|
2
|
-
import { InferSchema, FetchFunction } from '@ai-sdk/provider-utils';
|
|
3
|
-
import { ProviderV3, LanguageModelV3, ImageModelV3, EmbeddingModelV3, Experimental_VideoModelV3 } from '@ai-sdk/provider';
|
|
2
|
+
import { InferSchema, Resolvable, FetchFunction } from '@ai-sdk/provider-utils';
|
|
3
|
+
import { TranscriptionModelV3, ProviderV3, LanguageModelV3, ImageModelV3, EmbeddingModelV3, Experimental_VideoModelV3 } from '@ai-sdk/provider';
|
|
4
|
+
import { z } from 'zod/v4';
|
|
4
5
|
|
|
5
6
|
declare const googleErrorDataSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
6
7
|
error: {
|
|
@@ -54,7 +55,8 @@ declare const googleLanguageModelOptions: _ai_sdk_provider_utils.LazySchema<{
|
|
|
54
55
|
type GoogleLanguageModelOptions = InferSchema<typeof googleLanguageModelOptions>;
|
|
55
56
|
|
|
56
57
|
declare const responseSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
57
|
-
|
|
58
|
+
responseId?: string | null | undefined;
|
|
59
|
+
candidates?: {
|
|
58
60
|
content?: Record<string, never> | {
|
|
59
61
|
parts?: ({
|
|
60
62
|
functionCall: {
|
|
@@ -170,8 +172,7 @@ declare const responseSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
|
170
172
|
urlRetrievalStatus: string;
|
|
171
173
|
}[] | null | undefined;
|
|
172
174
|
} | null | undefined;
|
|
173
|
-
}[];
|
|
174
|
-
responseId?: string | null | undefined;
|
|
175
|
+
}[] | null | undefined;
|
|
175
176
|
usageMetadata?: {
|
|
176
177
|
cachedContentTokenCount?: number | null | undefined;
|
|
177
178
|
thoughtsTokenCount?: number | null | undefined;
|
|
@@ -201,9 +202,10 @@ declare const responseSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
|
201
202
|
}[] | null | undefined;
|
|
202
203
|
} | null | undefined;
|
|
203
204
|
}>;
|
|
204
|
-
type
|
|
205
|
-
type
|
|
206
|
-
type
|
|
205
|
+
type CandidateSchema = NonNullable<InferSchema<typeof responseSchema>['candidates']>[number];
|
|
206
|
+
type GroundingMetadataSchema = NonNullable<CandidateSchema['groundingMetadata']>;
|
|
207
|
+
type UrlContextMetadataSchema = NonNullable<CandidateSchema['urlContextMetadata']>;
|
|
208
|
+
type SafetyRatingSchema = NonNullable<CandidateSchema['safetyRatings']>[number];
|
|
207
209
|
type PromptFeedbackSchema = NonNullable<InferSchema<typeof responseSchema>['promptFeedback']>;
|
|
208
210
|
type UsageMetadataSchema = NonNullable<InferSchema<typeof responseSchema>['usageMetadata']>;
|
|
209
211
|
|
|
@@ -406,6 +408,50 @@ type GoogleInteractionsProviderMetadata = {
|
|
|
406
408
|
*/
|
|
407
409
|
type GoogleInteractionsAgentName = 'deep-research-pro-preview-12-2025' | 'deep-research-preview-04-2026' | 'deep-research-max-preview-04-2026' | 'antigravity-preview-05-2026';
|
|
408
410
|
|
|
411
|
+
type GoogleTranscriptionModelId = 'gemini-3.5-transcribe' | 'gemini-3.5-transcribe-live' | (string & {});
|
|
412
|
+
/**
|
|
413
|
+
* Speech recognition options for Gemini transcription models
|
|
414
|
+
* (`gemini-3.5-transcribe`). Maps onto Google's `AudioTranscriptionConfig`.
|
|
415
|
+
* The live variant (`gemini-3.5-transcribe-live`) requires streaming
|
|
416
|
+
* transcription, which is only available in AI SDK v7.
|
|
417
|
+
*/
|
|
418
|
+
declare const googleTranscriptionModelOptions: z.ZodObject<{
|
|
419
|
+
languageCodes: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
420
|
+
customVocabulary: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
421
|
+
wordTimestamp: z.ZodOptional<z.ZodBoolean>;
|
|
422
|
+
diarization: z.ZodOptional<z.ZodBoolean>;
|
|
423
|
+
mode: z.ZodOptional<z.ZodEnum<{
|
|
424
|
+
SMART: "SMART";
|
|
425
|
+
VERBATIM: "VERBATIM";
|
|
426
|
+
}>>;
|
|
427
|
+
}, z.core.$strip>;
|
|
428
|
+
type GoogleTranscriptionModelOptions = z.infer<typeof googleTranscriptionModelOptions>;
|
|
429
|
+
|
|
430
|
+
interface GoogleTranscriptionModelConfig {
|
|
431
|
+
provider: string;
|
|
432
|
+
baseURL: string;
|
|
433
|
+
headers?: Resolvable<Record<string, string | undefined>>;
|
|
434
|
+
fetch?: FetchFunction;
|
|
435
|
+
_internal?: {
|
|
436
|
+
currentDate?: () => Date;
|
|
437
|
+
};
|
|
438
|
+
}
|
|
439
|
+
/**
|
|
440
|
+
* Gemini transcription (speech-to-text) via the Interactions API
|
|
441
|
+
* (e.g. `gemini-3.5-transcribe`).
|
|
442
|
+
*
|
|
443
|
+
* @see https://ai.google.dev/gemini-api/docs/transcribe
|
|
444
|
+
*/
|
|
445
|
+
declare class GoogleTranscriptionModel implements TranscriptionModelV3 {
|
|
446
|
+
readonly modelId: GoogleTranscriptionModelId;
|
|
447
|
+
private readonly config;
|
|
448
|
+
readonly specificationVersion = "v3";
|
|
449
|
+
get provider(): string;
|
|
450
|
+
constructor(modelId: GoogleTranscriptionModelId, config: GoogleTranscriptionModelConfig);
|
|
451
|
+
private parseOptions;
|
|
452
|
+
doGenerate(options: Parameters<TranscriptionModelV3['doGenerate']>[0]): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>>;
|
|
453
|
+
}
|
|
454
|
+
|
|
409
455
|
declare const googleTools: {
|
|
410
456
|
/**
|
|
411
457
|
* Creates a Google search tool that gives Google direct access to real-time web content.
|
|
@@ -517,6 +563,15 @@ interface GoogleGenerativeAIProvider extends ProviderV3 {
|
|
|
517
563
|
* @deprecated Use `embeddingModel` instead.
|
|
518
564
|
*/
|
|
519
565
|
textEmbeddingModel(modelId: GoogleGenerativeAIEmbeddingModelId): EmbeddingModelV3;
|
|
566
|
+
/**
|
|
567
|
+
* Creates a model for transcription (speech-to-text), e.g.
|
|
568
|
+
* `gemini-3.5-transcribe`.
|
|
569
|
+
*/
|
|
570
|
+
transcription(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
571
|
+
/**
|
|
572
|
+
* Creates a model for transcription (speech-to-text).
|
|
573
|
+
*/
|
|
574
|
+
transcriptionModel(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
520
575
|
/**
|
|
521
576
|
* Creates a model for video generation.
|
|
522
577
|
*/
|
|
@@ -581,4 +636,4 @@ declare const google: GoogleGenerativeAIProvider;
|
|
|
581
636
|
|
|
582
637
|
declare const VERSION: string;
|
|
583
638
|
|
|
584
|
-
export { type GoogleEmbeddingModelOptions, type GoogleErrorData, type GoogleEmbeddingModelOptions as GoogleGenerativeAIEmbeddingProviderOptions, type GoogleImageModelOptions as GoogleGenerativeAIImageProviderOptions, type GoogleGenerativeAIProvider, type GoogleGenerativeAIProviderMetadata, type GoogleLanguageModelOptions as GoogleGenerativeAIProviderOptions, type GoogleGenerativeAIProviderSettings, type GoogleGenerativeAIVideoModelId, type GoogleVideoModelOptions as GoogleGenerativeAIVideoProviderOptions, type GoogleImageModelOptions, type GoogleInteractionsAgentName, type GoogleInteractionsModelId, type GoogleInteractionsProviderMetadata, type GoogleLanguageModelInteractionsOptions, type GoogleLanguageModelOptions, type GoogleVideoModelOptions, VERSION, createGoogleGenerativeAI, google };
|
|
639
|
+
export { type GoogleEmbeddingModelOptions, type GoogleErrorData, type GoogleEmbeddingModelOptions as GoogleGenerativeAIEmbeddingProviderOptions, type GoogleImageModelOptions as GoogleGenerativeAIImageProviderOptions, type GoogleGenerativeAIProvider, type GoogleGenerativeAIProviderMetadata, type GoogleLanguageModelOptions as GoogleGenerativeAIProviderOptions, type GoogleGenerativeAIProviderSettings, type GoogleGenerativeAIVideoModelId, type GoogleVideoModelOptions as GoogleGenerativeAIVideoProviderOptions, type GoogleImageModelOptions, type GoogleInteractionsAgentName, type GoogleInteractionsModelId, type GoogleInteractionsProviderMetadata, type GoogleLanguageModelInteractionsOptions, type GoogleLanguageModelOptions, GoogleTranscriptionModel, type GoogleTranscriptionModelId, type GoogleTranscriptionModelOptions, type GoogleVideoModelOptions, VERSION, createGoogleGenerativeAI, google };
|
package/dist/index.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import * as _ai_sdk_provider_utils from '@ai-sdk/provider-utils';
|
|
2
|
-
import { InferSchema, FetchFunction } from '@ai-sdk/provider-utils';
|
|
3
|
-
import { ProviderV3, LanguageModelV3, ImageModelV3, EmbeddingModelV3, Experimental_VideoModelV3 } from '@ai-sdk/provider';
|
|
2
|
+
import { InferSchema, Resolvable, FetchFunction } from '@ai-sdk/provider-utils';
|
|
3
|
+
import { TranscriptionModelV3, ProviderV3, LanguageModelV3, ImageModelV3, EmbeddingModelV3, Experimental_VideoModelV3 } from '@ai-sdk/provider';
|
|
4
|
+
import { z } from 'zod/v4';
|
|
4
5
|
|
|
5
6
|
declare const googleErrorDataSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
6
7
|
error: {
|
|
@@ -54,7 +55,8 @@ declare const googleLanguageModelOptions: _ai_sdk_provider_utils.LazySchema<{
|
|
|
54
55
|
type GoogleLanguageModelOptions = InferSchema<typeof googleLanguageModelOptions>;
|
|
55
56
|
|
|
56
57
|
declare const responseSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
57
|
-
|
|
58
|
+
responseId?: string | null | undefined;
|
|
59
|
+
candidates?: {
|
|
58
60
|
content?: Record<string, never> | {
|
|
59
61
|
parts?: ({
|
|
60
62
|
functionCall: {
|
|
@@ -170,8 +172,7 @@ declare const responseSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
|
170
172
|
urlRetrievalStatus: string;
|
|
171
173
|
}[] | null | undefined;
|
|
172
174
|
} | null | undefined;
|
|
173
|
-
}[];
|
|
174
|
-
responseId?: string | null | undefined;
|
|
175
|
+
}[] | null | undefined;
|
|
175
176
|
usageMetadata?: {
|
|
176
177
|
cachedContentTokenCount?: number | null | undefined;
|
|
177
178
|
thoughtsTokenCount?: number | null | undefined;
|
|
@@ -201,9 +202,10 @@ declare const responseSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
|
201
202
|
}[] | null | undefined;
|
|
202
203
|
} | null | undefined;
|
|
203
204
|
}>;
|
|
204
|
-
type
|
|
205
|
-
type
|
|
206
|
-
type
|
|
205
|
+
type CandidateSchema = NonNullable<InferSchema<typeof responseSchema>['candidates']>[number];
|
|
206
|
+
type GroundingMetadataSchema = NonNullable<CandidateSchema['groundingMetadata']>;
|
|
207
|
+
type UrlContextMetadataSchema = NonNullable<CandidateSchema['urlContextMetadata']>;
|
|
208
|
+
type SafetyRatingSchema = NonNullable<CandidateSchema['safetyRatings']>[number];
|
|
207
209
|
type PromptFeedbackSchema = NonNullable<InferSchema<typeof responseSchema>['promptFeedback']>;
|
|
208
210
|
type UsageMetadataSchema = NonNullable<InferSchema<typeof responseSchema>['usageMetadata']>;
|
|
209
211
|
|
|
@@ -406,6 +408,50 @@ type GoogleInteractionsProviderMetadata = {
|
|
|
406
408
|
*/
|
|
407
409
|
type GoogleInteractionsAgentName = 'deep-research-pro-preview-12-2025' | 'deep-research-preview-04-2026' | 'deep-research-max-preview-04-2026' | 'antigravity-preview-05-2026';
|
|
408
410
|
|
|
411
|
+
type GoogleTranscriptionModelId = 'gemini-3.5-transcribe' | 'gemini-3.5-transcribe-live' | (string & {});
|
|
412
|
+
/**
|
|
413
|
+
* Speech recognition options for Gemini transcription models
|
|
414
|
+
* (`gemini-3.5-transcribe`). Maps onto Google's `AudioTranscriptionConfig`.
|
|
415
|
+
* The live variant (`gemini-3.5-transcribe-live`) requires streaming
|
|
416
|
+
* transcription, which is only available in AI SDK v7.
|
|
417
|
+
*/
|
|
418
|
+
declare const googleTranscriptionModelOptions: z.ZodObject<{
|
|
419
|
+
languageCodes: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
420
|
+
customVocabulary: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
421
|
+
wordTimestamp: z.ZodOptional<z.ZodBoolean>;
|
|
422
|
+
diarization: z.ZodOptional<z.ZodBoolean>;
|
|
423
|
+
mode: z.ZodOptional<z.ZodEnum<{
|
|
424
|
+
SMART: "SMART";
|
|
425
|
+
VERBATIM: "VERBATIM";
|
|
426
|
+
}>>;
|
|
427
|
+
}, z.core.$strip>;
|
|
428
|
+
type GoogleTranscriptionModelOptions = z.infer<typeof googleTranscriptionModelOptions>;
|
|
429
|
+
|
|
430
|
+
interface GoogleTranscriptionModelConfig {
|
|
431
|
+
provider: string;
|
|
432
|
+
baseURL: string;
|
|
433
|
+
headers?: Resolvable<Record<string, string | undefined>>;
|
|
434
|
+
fetch?: FetchFunction;
|
|
435
|
+
_internal?: {
|
|
436
|
+
currentDate?: () => Date;
|
|
437
|
+
};
|
|
438
|
+
}
|
|
439
|
+
/**
|
|
440
|
+
* Gemini transcription (speech-to-text) via the Interactions API
|
|
441
|
+
* (e.g. `gemini-3.5-transcribe`).
|
|
442
|
+
*
|
|
443
|
+
* @see https://ai.google.dev/gemini-api/docs/transcribe
|
|
444
|
+
*/
|
|
445
|
+
declare class GoogleTranscriptionModel implements TranscriptionModelV3 {
|
|
446
|
+
readonly modelId: GoogleTranscriptionModelId;
|
|
447
|
+
private readonly config;
|
|
448
|
+
readonly specificationVersion = "v3";
|
|
449
|
+
get provider(): string;
|
|
450
|
+
constructor(modelId: GoogleTranscriptionModelId, config: GoogleTranscriptionModelConfig);
|
|
451
|
+
private parseOptions;
|
|
452
|
+
doGenerate(options: Parameters<TranscriptionModelV3['doGenerate']>[0]): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>>;
|
|
453
|
+
}
|
|
454
|
+
|
|
409
455
|
declare const googleTools: {
|
|
410
456
|
/**
|
|
411
457
|
* Creates a Google search tool that gives Google direct access to real-time web content.
|
|
@@ -517,6 +563,15 @@ interface GoogleGenerativeAIProvider extends ProviderV3 {
|
|
|
517
563
|
* @deprecated Use `embeddingModel` instead.
|
|
518
564
|
*/
|
|
519
565
|
textEmbeddingModel(modelId: GoogleGenerativeAIEmbeddingModelId): EmbeddingModelV3;
|
|
566
|
+
/**
|
|
567
|
+
* Creates a model for transcription (speech-to-text), e.g.
|
|
568
|
+
* `gemini-3.5-transcribe`.
|
|
569
|
+
*/
|
|
570
|
+
transcription(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
571
|
+
/**
|
|
572
|
+
* Creates a model for transcription (speech-to-text).
|
|
573
|
+
*/
|
|
574
|
+
transcriptionModel(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
520
575
|
/**
|
|
521
576
|
* Creates a model for video generation.
|
|
522
577
|
*/
|
|
@@ -581,4 +636,4 @@ declare const google: GoogleGenerativeAIProvider;
|
|
|
581
636
|
|
|
582
637
|
declare const VERSION: string;
|
|
583
638
|
|
|
584
|
-
export { type GoogleEmbeddingModelOptions, type GoogleErrorData, type GoogleEmbeddingModelOptions as GoogleGenerativeAIEmbeddingProviderOptions, type GoogleImageModelOptions as GoogleGenerativeAIImageProviderOptions, type GoogleGenerativeAIProvider, type GoogleGenerativeAIProviderMetadata, type GoogleLanguageModelOptions as GoogleGenerativeAIProviderOptions, type GoogleGenerativeAIProviderSettings, type GoogleGenerativeAIVideoModelId, type GoogleVideoModelOptions as GoogleGenerativeAIVideoProviderOptions, type GoogleImageModelOptions, type GoogleInteractionsAgentName, type GoogleInteractionsModelId, type GoogleInteractionsProviderMetadata, type GoogleLanguageModelInteractionsOptions, type GoogleLanguageModelOptions, type GoogleVideoModelOptions, VERSION, createGoogleGenerativeAI, google };
|
|
639
|
+
export { type GoogleEmbeddingModelOptions, type GoogleErrorData, type GoogleEmbeddingModelOptions as GoogleGenerativeAIEmbeddingProviderOptions, type GoogleImageModelOptions as GoogleGenerativeAIImageProviderOptions, type GoogleGenerativeAIProvider, type GoogleGenerativeAIProviderMetadata, type GoogleLanguageModelOptions as GoogleGenerativeAIProviderOptions, type GoogleGenerativeAIProviderSettings, type GoogleGenerativeAIVideoModelId, type GoogleVideoModelOptions as GoogleGenerativeAIVideoProviderOptions, type GoogleImageModelOptions, type GoogleInteractionsAgentName, type GoogleInteractionsModelId, type GoogleInteractionsProviderMetadata, type GoogleLanguageModelInteractionsOptions, type GoogleLanguageModelOptions, GoogleTranscriptionModel, type GoogleTranscriptionModelId, type GoogleTranscriptionModelOptions, type GoogleVideoModelOptions, VERSION, createGoogleGenerativeAI, google };
|