@ai-sdk/google 3.0.113 → 3.0.116
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/dist/index.d.mts +57 -3
- package/dist/index.d.ts +57 -3
- package/dist/index.js +953 -683
- package/dist/index.js.map +1 -1
- package/dist/index.mjs +891 -613
- package/dist/index.mjs.map +1 -1
- package/dist/internal/index.js +93 -17
- package/dist/internal/index.js.map +1 -1
- package/dist/internal/index.mjs +93 -17
- package/dist/internal/index.mjs.map +1 -1
- package/package.json +2 -2
- package/src/convert-json-schema-to-openapi-schema.ts +136 -8
- package/src/google-prepare-tools.ts +45 -21
- package/src/google-provider.ts +24 -0
- package/src/index.ts +5 -0
- package/src/transcription/google-transcription-model-options.ts +51 -0
- package/src/transcription/google-transcription-model.ts +243 -0
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,25 @@
|
|
|
1
1
|
# @ai-sdk/google
|
|
2
2
|
|
|
3
|
+
## 3.0.116
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- 9a1656e: fix(google): convert enum values to the Gemini schema format
|
|
8
|
+
- cfdc8df: feat (provider/google, provider/google-vertex): Gemini 3.5 Transcribe support — unary transcription (`gemini-3.5-transcribe`) via generateContent with language detection, speaker diarization, word timestamps, custom vocabulary, and `mode: 'VERBATIM' | 'SMART'` transcription formatting
|
|
9
|
+
|
|
10
|
+
## 3.0.115
|
|
11
|
+
|
|
12
|
+
### Patch Changes
|
|
13
|
+
|
|
14
|
+
- Updated dependencies [9a521b9]
|
|
15
|
+
- @ai-sdk/provider-utils@4.0.49
|
|
16
|
+
|
|
17
|
+
## 3.0.114
|
|
18
|
+
|
|
19
|
+
### Patch Changes
|
|
20
|
+
|
|
21
|
+
- 1df2ced: Preserve recursive tool input schemas without aborting Google model calls.
|
|
22
|
+
|
|
3
23
|
## 3.0.113
|
|
4
24
|
|
|
5
25
|
### Patch Changes
|
package/dist/index.d.mts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import * as _ai_sdk_provider_utils from '@ai-sdk/provider-utils';
|
|
2
|
-
import { InferSchema, FetchFunction } from '@ai-sdk/provider-utils';
|
|
3
|
-
import { ProviderV3, LanguageModelV3, ImageModelV3, EmbeddingModelV3, Experimental_VideoModelV3 } from '@ai-sdk/provider';
|
|
2
|
+
import { InferSchema, Resolvable, FetchFunction } from '@ai-sdk/provider-utils';
|
|
3
|
+
import { TranscriptionModelV3, ProviderV3, LanguageModelV3, ImageModelV3, EmbeddingModelV3, Experimental_VideoModelV3 } from '@ai-sdk/provider';
|
|
4
|
+
import { z } from 'zod/v4';
|
|
4
5
|
|
|
5
6
|
declare const googleErrorDataSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
6
7
|
error: {
|
|
@@ -406,6 +407,50 @@ type GoogleInteractionsProviderMetadata = {
|
|
|
406
407
|
*/
|
|
407
408
|
type GoogleInteractionsAgentName = 'deep-research-pro-preview-12-2025' | 'deep-research-preview-04-2026' | 'deep-research-max-preview-04-2026' | 'antigravity-preview-05-2026';
|
|
408
409
|
|
|
410
|
+
type GoogleTranscriptionModelId = 'gemini-3.5-transcribe' | 'gemini-3.5-transcribe-live' | (string & {});
|
|
411
|
+
/**
|
|
412
|
+
* Speech recognition options for Gemini transcription models
|
|
413
|
+
* (`gemini-3.5-transcribe`). Maps onto Google's `AudioTranscriptionConfig`.
|
|
414
|
+
* The live variant (`gemini-3.5-transcribe-live`) requires streaming
|
|
415
|
+
* transcription, which is only available in AI SDK v7.
|
|
416
|
+
*/
|
|
417
|
+
declare const googleTranscriptionModelOptions: z.ZodObject<{
|
|
418
|
+
languageCodes: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
419
|
+
customVocabulary: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
420
|
+
wordTimestamp: z.ZodOptional<z.ZodBoolean>;
|
|
421
|
+
diarization: z.ZodOptional<z.ZodBoolean>;
|
|
422
|
+
mode: z.ZodOptional<z.ZodEnum<{
|
|
423
|
+
SMART: "SMART";
|
|
424
|
+
VERBATIM: "VERBATIM";
|
|
425
|
+
}>>;
|
|
426
|
+
}, z.core.$strip>;
|
|
427
|
+
type GoogleTranscriptionModelOptions = z.infer<typeof googleTranscriptionModelOptions>;
|
|
428
|
+
|
|
429
|
+
interface GoogleTranscriptionModelConfig {
|
|
430
|
+
provider: string;
|
|
431
|
+
baseURL: string;
|
|
432
|
+
headers?: Resolvable<Record<string, string | undefined>>;
|
|
433
|
+
fetch?: FetchFunction;
|
|
434
|
+
_internal?: {
|
|
435
|
+
currentDate?: () => Date;
|
|
436
|
+
};
|
|
437
|
+
}
|
|
438
|
+
/**
|
|
439
|
+
* Gemini transcription (speech-to-text) via the Interactions API
|
|
440
|
+
* (e.g. `gemini-3.5-transcribe`).
|
|
441
|
+
*
|
|
442
|
+
* @see https://ai.google.dev/gemini-api/docs/transcribe
|
|
443
|
+
*/
|
|
444
|
+
declare class GoogleTranscriptionModel implements TranscriptionModelV3 {
|
|
445
|
+
readonly modelId: GoogleTranscriptionModelId;
|
|
446
|
+
private readonly config;
|
|
447
|
+
readonly specificationVersion = "v3";
|
|
448
|
+
get provider(): string;
|
|
449
|
+
constructor(modelId: GoogleTranscriptionModelId, config: GoogleTranscriptionModelConfig);
|
|
450
|
+
private parseOptions;
|
|
451
|
+
doGenerate(options: Parameters<TranscriptionModelV3['doGenerate']>[0]): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>>;
|
|
452
|
+
}
|
|
453
|
+
|
|
409
454
|
declare const googleTools: {
|
|
410
455
|
/**
|
|
411
456
|
* Creates a Google search tool that gives Google direct access to real-time web content.
|
|
@@ -517,6 +562,15 @@ interface GoogleGenerativeAIProvider extends ProviderV3 {
|
|
|
517
562
|
* @deprecated Use `embeddingModel` instead.
|
|
518
563
|
*/
|
|
519
564
|
textEmbeddingModel(modelId: GoogleGenerativeAIEmbeddingModelId): EmbeddingModelV3;
|
|
565
|
+
/**
|
|
566
|
+
* Creates a model for transcription (speech-to-text), e.g.
|
|
567
|
+
* `gemini-3.5-transcribe`.
|
|
568
|
+
*/
|
|
569
|
+
transcription(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
570
|
+
/**
|
|
571
|
+
* Creates a model for transcription (speech-to-text).
|
|
572
|
+
*/
|
|
573
|
+
transcriptionModel(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
520
574
|
/**
|
|
521
575
|
* Creates a model for video generation.
|
|
522
576
|
*/
|
|
@@ -581,4 +635,4 @@ declare const google: GoogleGenerativeAIProvider;
|
|
|
581
635
|
|
|
582
636
|
declare const VERSION: string;
|
|
583
637
|
|
|
584
|
-
export { type GoogleEmbeddingModelOptions, type GoogleErrorData, type GoogleEmbeddingModelOptions as GoogleGenerativeAIEmbeddingProviderOptions, type GoogleImageModelOptions as GoogleGenerativeAIImageProviderOptions, type GoogleGenerativeAIProvider, type GoogleGenerativeAIProviderMetadata, type GoogleLanguageModelOptions as GoogleGenerativeAIProviderOptions, type GoogleGenerativeAIProviderSettings, type GoogleGenerativeAIVideoModelId, type GoogleVideoModelOptions as GoogleGenerativeAIVideoProviderOptions, type GoogleImageModelOptions, type GoogleInteractionsAgentName, type GoogleInteractionsModelId, type GoogleInteractionsProviderMetadata, type GoogleLanguageModelInteractionsOptions, type GoogleLanguageModelOptions, type GoogleVideoModelOptions, VERSION, createGoogleGenerativeAI, google };
|
|
638
|
+
export { type GoogleEmbeddingModelOptions, type GoogleErrorData, type GoogleEmbeddingModelOptions as GoogleGenerativeAIEmbeddingProviderOptions, type GoogleImageModelOptions as GoogleGenerativeAIImageProviderOptions, type GoogleGenerativeAIProvider, type GoogleGenerativeAIProviderMetadata, type GoogleLanguageModelOptions as GoogleGenerativeAIProviderOptions, type GoogleGenerativeAIProviderSettings, type GoogleGenerativeAIVideoModelId, type GoogleVideoModelOptions as GoogleGenerativeAIVideoProviderOptions, type GoogleImageModelOptions, type GoogleInteractionsAgentName, type GoogleInteractionsModelId, type GoogleInteractionsProviderMetadata, type GoogleLanguageModelInteractionsOptions, type GoogleLanguageModelOptions, GoogleTranscriptionModel, type GoogleTranscriptionModelId, type GoogleTranscriptionModelOptions, type GoogleVideoModelOptions, VERSION, createGoogleGenerativeAI, google };
|
package/dist/index.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import * as _ai_sdk_provider_utils from '@ai-sdk/provider-utils';
|
|
2
|
-
import { InferSchema, FetchFunction } from '@ai-sdk/provider-utils';
|
|
3
|
-
import { ProviderV3, LanguageModelV3, ImageModelV3, EmbeddingModelV3, Experimental_VideoModelV3 } from '@ai-sdk/provider';
|
|
2
|
+
import { InferSchema, Resolvable, FetchFunction } from '@ai-sdk/provider-utils';
|
|
3
|
+
import { TranscriptionModelV3, ProviderV3, LanguageModelV3, ImageModelV3, EmbeddingModelV3, Experimental_VideoModelV3 } from '@ai-sdk/provider';
|
|
4
|
+
import { z } from 'zod/v4';
|
|
4
5
|
|
|
5
6
|
declare const googleErrorDataSchema: _ai_sdk_provider_utils.LazySchema<{
|
|
6
7
|
error: {
|
|
@@ -406,6 +407,50 @@ type GoogleInteractionsProviderMetadata = {
|
|
|
406
407
|
*/
|
|
407
408
|
type GoogleInteractionsAgentName = 'deep-research-pro-preview-12-2025' | 'deep-research-preview-04-2026' | 'deep-research-max-preview-04-2026' | 'antigravity-preview-05-2026';
|
|
408
409
|
|
|
410
|
+
type GoogleTranscriptionModelId = 'gemini-3.5-transcribe' | 'gemini-3.5-transcribe-live' | (string & {});
|
|
411
|
+
/**
|
|
412
|
+
* Speech recognition options for Gemini transcription models
|
|
413
|
+
* (`gemini-3.5-transcribe`). Maps onto Google's `AudioTranscriptionConfig`.
|
|
414
|
+
* The live variant (`gemini-3.5-transcribe-live`) requires streaming
|
|
415
|
+
* transcription, which is only available in AI SDK v7.
|
|
416
|
+
*/
|
|
417
|
+
declare const googleTranscriptionModelOptions: z.ZodObject<{
|
|
418
|
+
languageCodes: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
419
|
+
customVocabulary: z.ZodOptional<z.ZodArray<z.ZodString>>;
|
|
420
|
+
wordTimestamp: z.ZodOptional<z.ZodBoolean>;
|
|
421
|
+
diarization: z.ZodOptional<z.ZodBoolean>;
|
|
422
|
+
mode: z.ZodOptional<z.ZodEnum<{
|
|
423
|
+
SMART: "SMART";
|
|
424
|
+
VERBATIM: "VERBATIM";
|
|
425
|
+
}>>;
|
|
426
|
+
}, z.core.$strip>;
|
|
427
|
+
type GoogleTranscriptionModelOptions = z.infer<typeof googleTranscriptionModelOptions>;
|
|
428
|
+
|
|
429
|
+
interface GoogleTranscriptionModelConfig {
|
|
430
|
+
provider: string;
|
|
431
|
+
baseURL: string;
|
|
432
|
+
headers?: Resolvable<Record<string, string | undefined>>;
|
|
433
|
+
fetch?: FetchFunction;
|
|
434
|
+
_internal?: {
|
|
435
|
+
currentDate?: () => Date;
|
|
436
|
+
};
|
|
437
|
+
}
|
|
438
|
+
/**
|
|
439
|
+
* Gemini transcription (speech-to-text) via the Interactions API
|
|
440
|
+
* (e.g. `gemini-3.5-transcribe`).
|
|
441
|
+
*
|
|
442
|
+
* @see https://ai.google.dev/gemini-api/docs/transcribe
|
|
443
|
+
*/
|
|
444
|
+
declare class GoogleTranscriptionModel implements TranscriptionModelV3 {
|
|
445
|
+
readonly modelId: GoogleTranscriptionModelId;
|
|
446
|
+
private readonly config;
|
|
447
|
+
readonly specificationVersion = "v3";
|
|
448
|
+
get provider(): string;
|
|
449
|
+
constructor(modelId: GoogleTranscriptionModelId, config: GoogleTranscriptionModelConfig);
|
|
450
|
+
private parseOptions;
|
|
451
|
+
doGenerate(options: Parameters<TranscriptionModelV3['doGenerate']>[0]): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>>;
|
|
452
|
+
}
|
|
453
|
+
|
|
409
454
|
declare const googleTools: {
|
|
410
455
|
/**
|
|
411
456
|
* Creates a Google search tool that gives Google direct access to real-time web content.
|
|
@@ -517,6 +562,15 @@ interface GoogleGenerativeAIProvider extends ProviderV3 {
|
|
|
517
562
|
* @deprecated Use `embeddingModel` instead.
|
|
518
563
|
*/
|
|
519
564
|
textEmbeddingModel(modelId: GoogleGenerativeAIEmbeddingModelId): EmbeddingModelV3;
|
|
565
|
+
/**
|
|
566
|
+
* Creates a model for transcription (speech-to-text), e.g.
|
|
567
|
+
* `gemini-3.5-transcribe`.
|
|
568
|
+
*/
|
|
569
|
+
transcription(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
570
|
+
/**
|
|
571
|
+
* Creates a model for transcription (speech-to-text).
|
|
572
|
+
*/
|
|
573
|
+
transcriptionModel(modelId: GoogleTranscriptionModelId): TranscriptionModelV3;
|
|
520
574
|
/**
|
|
521
575
|
* Creates a model for video generation.
|
|
522
576
|
*/
|
|
@@ -581,4 +635,4 @@ declare const google: GoogleGenerativeAIProvider;
|
|
|
581
635
|
|
|
582
636
|
declare const VERSION: string;
|
|
583
637
|
|
|
584
|
-
export { type GoogleEmbeddingModelOptions, type GoogleErrorData, type GoogleEmbeddingModelOptions as GoogleGenerativeAIEmbeddingProviderOptions, type GoogleImageModelOptions as GoogleGenerativeAIImageProviderOptions, type GoogleGenerativeAIProvider, type GoogleGenerativeAIProviderMetadata, type GoogleLanguageModelOptions as GoogleGenerativeAIProviderOptions, type GoogleGenerativeAIProviderSettings, type GoogleGenerativeAIVideoModelId, type GoogleVideoModelOptions as GoogleGenerativeAIVideoProviderOptions, type GoogleImageModelOptions, type GoogleInteractionsAgentName, type GoogleInteractionsModelId, type GoogleInteractionsProviderMetadata, type GoogleLanguageModelInteractionsOptions, type GoogleLanguageModelOptions, type GoogleVideoModelOptions, VERSION, createGoogleGenerativeAI, google };
|
|
638
|
+
export { type GoogleEmbeddingModelOptions, type GoogleErrorData, type GoogleEmbeddingModelOptions as GoogleGenerativeAIEmbeddingProviderOptions, type GoogleImageModelOptions as GoogleGenerativeAIImageProviderOptions, type GoogleGenerativeAIProvider, type GoogleGenerativeAIProviderMetadata, type GoogleLanguageModelOptions as GoogleGenerativeAIProviderOptions, type GoogleGenerativeAIProviderSettings, type GoogleGenerativeAIVideoModelId, type GoogleVideoModelOptions as GoogleGenerativeAIVideoProviderOptions, type GoogleImageModelOptions, type GoogleInteractionsAgentName, type GoogleInteractionsModelId, type GoogleInteractionsProviderMetadata, type GoogleLanguageModelInteractionsOptions, type GoogleLanguageModelOptions, GoogleTranscriptionModel, type GoogleTranscriptionModelId, type GoogleTranscriptionModelOptions, type GoogleVideoModelOptions, VERSION, createGoogleGenerativeAI, google };
|