@ai-sdk/google 3.0.114 → 3.0.118

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,243 @@
1
+ import {
2
+ InvalidArgumentError,
3
+ type JSONObject,
4
+ type SharedV3Warning,
5
+ type TranscriptionModelV3,
6
+ } from '@ai-sdk/provider';
7
+ import {
8
+ combineHeaders,
9
+ convertToBase64,
10
+ createJsonResponseHandler,
11
+ parseProviderOptions,
12
+ postJsonToApi,
13
+ resolve,
14
+ type FetchFunction,
15
+ type Resolvable,
16
+ } from '@ai-sdk/provider-utils';
17
+ import { z } from 'zod/v4';
18
+ import { googleFailedResponseHandler } from '../google-error';
19
+ import {
20
+ googleTranscriptionModelOptions,
21
+ type GoogleTranscriptionModelId,
22
+ type GoogleTranscriptionModelOptions,
23
+ } from './google-transcription-model-options';
24
+
25
+ /**
26
+ * Live transcription (`*-live` model variants) requires streaming support,
27
+ * which is only available in AI SDK v7 (transcription specification v4).
28
+ */
29
+ function isLiveTranscriptionModelId(modelId: string): boolean {
30
+ return modelId.includes('-live');
31
+ }
32
+
33
+ interface GoogleTranscriptionModelConfig {
34
+ provider: string;
35
+ baseURL: string;
36
+ headers?: Resolvable<Record<string, string | undefined>>;
37
+ fetch?: FetchFunction;
38
+ _internal?: {
39
+ currentDate?: () => Date;
40
+ };
41
+ }
42
+
43
+ /**
44
+ * Gemini transcription (speech-to-text) via the Interactions API
45
+ * (e.g. `gemini-3.5-transcribe`).
46
+ *
47
+ * @see https://ai.google.dev/gemini-api/docs/transcribe
48
+ */
49
+ export class GoogleTranscriptionModel implements TranscriptionModelV3 {
50
+ readonly specificationVersion = 'v3';
51
+
52
+ get provider(): string {
53
+ return this.config.provider;
54
+ }
55
+
56
+ constructor(
57
+ readonly modelId: GoogleTranscriptionModelId,
58
+ private readonly config: GoogleTranscriptionModelConfig,
59
+ ) {}
60
+
61
+ private async parseOptions(
62
+ providerOptions: Record<string, unknown> | undefined,
63
+ ): Promise<GoogleTranscriptionModelOptions | undefined> {
64
+ return parseProviderOptions({
65
+ provider: 'google',
66
+ providerOptions,
67
+ schema: googleTranscriptionModelOptions,
68
+ });
69
+ }
70
+
71
+ async doGenerate(
72
+ options: Parameters<TranscriptionModelV3['doGenerate']>[0],
73
+ ): Promise<Awaited<ReturnType<TranscriptionModelV3['doGenerate']>>> {
74
+ if (isLiveTranscriptionModelId(this.modelId)) {
75
+ throw new InvalidArgumentError({
76
+ argument: 'modelId',
77
+ message:
78
+ `Model '${this.modelId}' only supports streaming transcription, ` +
79
+ `which requires AI SDK v7. Use a unary model such as 'gemini-3.5-transcribe'.`,
80
+ });
81
+ }
82
+
83
+ const currentDate = this.config._internal?.currentDate?.() ?? new Date();
84
+ const warnings: SharedV3Warning[] = [];
85
+ const googleOptions = await this.parseOptions(options.providerOptions);
86
+ const transcriptionConfig = buildTranscriptionConfig(googleOptions);
87
+
88
+ // Unary transcription is served by the Interactions API
89
+ // (https://ai.google.dev/gemini-api/docs/transcribe).
90
+ const requestBody = {
91
+ model: this.modelId,
92
+ input: [
93
+ {
94
+ type: 'audio',
95
+ data: convertToBase64(options.audio),
96
+ mime_type: options.mediaType,
97
+ },
98
+ ],
99
+ ...(transcriptionConfig != null
100
+ ? { generation_config: { transcription_config: transcriptionConfig } }
101
+ : {}),
102
+ };
103
+
104
+ const {
105
+ value: response,
106
+ responseHeaders,
107
+ rawValue: rawResponse,
108
+ } = await postJsonToApi({
109
+ url: `${this.config.baseURL}/interactions`,
110
+ headers: combineHeaders(
111
+ this.config.headers ? await resolve(this.config.headers) : undefined,
112
+ options.headers,
113
+ ),
114
+ body: requestBody,
115
+ failedResponseHandler: googleFailedResponseHandler,
116
+ successfulResponseHandler: createJsonResponseHandler(
117
+ googleInteractionsTranscriptionResponseSchema,
118
+ ),
119
+ abortSignal: options.abortSignal,
120
+ fetch: this.config.fetch,
121
+ });
122
+
123
+ let text = '';
124
+ const segments: Array<{
125
+ text: string;
126
+ startSecond: number;
127
+ endSecond: number;
128
+ }> = [];
129
+ for (const step of response.steps ?? []) {
130
+ for (const content of step.content ?? []) {
131
+ if (content.type !== 'text' || content.text == null) continue;
132
+ text += content.text;
133
+ for (const annotation of content.annotations ?? []) {
134
+ if (annotation.type !== 'word_info') continue;
135
+ const startSecond = parseOffsetSeconds(annotation.start_offset);
136
+ const endSecond = parseOffsetSeconds(annotation.end_offset);
137
+ if (
138
+ annotation.text == null ||
139
+ startSecond == null ||
140
+ endSecond == null
141
+ ) {
142
+ continue;
143
+ }
144
+ segments.push({ text: annotation.text, startSecond, endSecond });
145
+ }
146
+ }
147
+ }
148
+
149
+ return {
150
+ text,
151
+ segments,
152
+ language: undefined,
153
+ durationInSeconds: undefined,
154
+ warnings,
155
+ response: {
156
+ timestamp: currentDate,
157
+ modelId: this.modelId,
158
+ headers: responseHeaders,
159
+ body: rawResponse,
160
+ },
161
+ ...(response.usage != null
162
+ ? {
163
+ providerMetadata: {
164
+ google: { usage: response.usage as JSONObject },
165
+ },
166
+ }
167
+ : {}),
168
+ };
169
+ }
170
+ }
171
+
172
+ /**
173
+ * Builds the Interactions API `transcription_config` (snake_case wire) from
174
+ * provider options; returns undefined when no options are set. Diarization
175
+ * and word timestamps are expressed inside the `mode` object per
176
+ * https://ai.google.dev/gemini-api/docs/transcribe.
177
+ */
178
+ function buildTranscriptionConfig(
179
+ options: GoogleTranscriptionModelOptions | undefined,
180
+ ): Record<string, unknown> | undefined {
181
+ if (options == null) return undefined;
182
+ const config: Record<string, unknown> = {};
183
+ if (options.languageCodes != null) {
184
+ config.language_codes = options.languageCodes;
185
+ }
186
+ if (options.customVocabulary != null) {
187
+ config.custom_vocabulary = options.customVocabulary;
188
+ }
189
+ if (
190
+ options.mode != null ||
191
+ options.diarization === true ||
192
+ options.wordTimestamp === true
193
+ ) {
194
+ config.mode = {
195
+ type: (options.mode ?? 'VERBATIM').toLowerCase(),
196
+ ...(options.diarization === true ? { diarization_mode: 'speaker' } : {}),
197
+ ...(options.wordTimestamp === true
198
+ ? { timestamp_granularities: ['word'] }
199
+ : {}),
200
+ };
201
+ }
202
+ return Object.keys(config).length > 0 ? config : undefined;
203
+ }
204
+
205
+ /** Parses a Google duration offset such as `"1s"` or `"9.400s"` to seconds. */
206
+ function parseOffsetSeconds(
207
+ offset: string | undefined | null,
208
+ ): number | undefined {
209
+ if (offset == null) return undefined;
210
+ const parsed = Number.parseFloat(offset);
211
+ return Number.isFinite(parsed) ? parsed : undefined;
212
+ }
213
+
214
+ const googleInteractionsWordAnnotationSchema = z.object({
215
+ type: z.string().nullish(),
216
+ text: z.string().nullish(),
217
+ speaker: z.string().nullish(),
218
+ start_offset: z.string().nullish(),
219
+ end_offset: z.string().nullish(),
220
+ });
221
+
222
+ const googleInteractionsTranscriptionResponseSchema = z.object({
223
+ status: z.string().nullish(),
224
+ steps: z
225
+ .array(
226
+ z.object({
227
+ type: z.string().nullish(),
228
+ content: z
229
+ .array(
230
+ z.object({
231
+ type: z.string().nullish(),
232
+ text: z.string().nullish(),
233
+ annotations: z
234
+ .array(googleInteractionsWordAnnotationSchema)
235
+ .nullish(),
236
+ }),
237
+ )
238
+ .nullish(),
239
+ }),
240
+ )
241
+ .nullish(),
242
+ usage: z.record(z.string(), z.unknown()).nullish(),
243
+ });