@ai-sdk/gateway 4.0.75 → 4.0.76
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/index.d.ts +10 -6
- package/dist/index.js +291 -244
- package/dist/index.js.map +1 -1
- package/docs/00-ai-gateway.mdx +23 -14
- package/package.json +3 -3
- package/src/{gateway-language-model-batch.ts → gateway-batch.ts} +86 -44
- package/src/gateway-language-model-settings.ts +4 -2
- package/src/gateway-provider-options.ts +1 -1
- package/src/gateway-provider.ts +21 -6
package/docs/00-ai-gateway.mdx
CHANGED
|
@@ -219,15 +219,15 @@ the failure without inspecting the Gateway's serialized error payload.
|
|
|
219
219
|
Text batch support is experimental and the API may change in patch releases.
|
|
220
220
|
</Note>
|
|
221
221
|
|
|
222
|
-
AI Gateway supports durable text batches through `
|
|
222
|
+
AI Gateway supports durable text batches through `experimental_startBatch`, `experimental_getBatchStatus`, and `experimental_getBatchResults`. See the [AI Gateway batch processing guide](https://vercel.com/docs/ai-gateway/models-and-providers/batch-processing) for supported models, limits, and the complete lifecycle.
|
|
223
223
|
|
|
224
224
|
Pass a publicly reachable HTTPS `webhookUrl` when starting a batch to receive a terminal notification instead of polling. AI Gateway sends `batch.completed`, `batch.failed`, or `batch.cancelled`. The event contains terminal status information but not batch results. Completion webhooks are an AI Gateway capability; direct Anthropic and OpenAI batch providers return an unsupported warning when this option is provided.
|
|
225
225
|
|
|
226
226
|
```ts filename="start-batch.ts"
|
|
227
227
|
import { randomUUID } from 'node:crypto';
|
|
228
228
|
import {
|
|
229
|
-
|
|
230
|
-
type
|
|
229
|
+
experimental_startBatch as startBatch,
|
|
230
|
+
type Experimental_BatchReference as BatchReference,
|
|
231
231
|
type GatewayProviderMetadata,
|
|
232
232
|
} from 'ai';
|
|
233
233
|
|
|
@@ -240,11 +240,20 @@ const model = 'anthropic/claude-haiku-4.5' as const;
|
|
|
240
240
|
// pending record rather than an unknown callback.
|
|
241
241
|
await reserveBatchWebhook(token);
|
|
242
242
|
|
|
243
|
-
const batch = await
|
|
244
|
-
model,
|
|
243
|
+
const batch = await startBatch({
|
|
245
244
|
requests: [
|
|
246
|
-
{
|
|
247
|
-
|
|
245
|
+
{
|
|
246
|
+
id: 'france',
|
|
247
|
+
type: 'text',
|
|
248
|
+
model,
|
|
249
|
+
prompt: 'What is the capital of France?',
|
|
250
|
+
},
|
|
251
|
+
{
|
|
252
|
+
id: 'germany',
|
|
253
|
+
type: 'text',
|
|
254
|
+
model,
|
|
255
|
+
prompt: 'What is the capital of Germany?',
|
|
256
|
+
},
|
|
248
257
|
],
|
|
249
258
|
providerOptions: {
|
|
250
259
|
gateway: { idempotencyKey: token },
|
|
@@ -264,20 +273,17 @@ if (typeof signingSecret !== 'string') {
|
|
|
264
273
|
// Keep sensitive provider metadata out of the receiver queue and logs.
|
|
265
274
|
const batchReference = {
|
|
266
275
|
version: batch.version,
|
|
267
|
-
type: batch.type,
|
|
268
276
|
id: batch.id,
|
|
269
277
|
provider: batch.provider,
|
|
270
|
-
|
|
271
|
-
} satisfies TextBatchReference;
|
|
278
|
+
} satisfies BatchReference;
|
|
272
279
|
|
|
273
280
|
await completeBatchWebhookReservation(token, {
|
|
274
281
|
batch: batchReference,
|
|
275
|
-
model,
|
|
276
282
|
signingSecret,
|
|
277
283
|
});
|
|
278
284
|
```
|
|
279
285
|
|
|
280
|
-
The signing secret is returned only in the start response. Reserve the callback token before submitting the batch, then persist the secret
|
|
286
|
+
The signing secret is returned only in the start response. Reserve the callback token before submitting the batch, then persist the secret and minimal `batch` reference before the process exits. The stable `providerOptions.gateway.idempotencyKey` makes retrying an ambiguous start safe. Reuse the reserved token when retrying; do not generate a new one. Delete the reservation after a definitive start failure. While the reservation is still pending, the receiver should return a retryable response so an early delivery is sent again after setup completes.
|
|
281
287
|
|
|
282
288
|
The receiver must verify the `x-ai-gateway-signature` header against a bounded raw request body before trusting the event:
|
|
283
289
|
|
|
@@ -340,7 +346,6 @@ export async function POST(request: Request) {
|
|
|
340
346
|
await enqueueBatchResultFetch({
|
|
341
347
|
batch: record.batch,
|
|
342
348
|
deliveryId,
|
|
343
|
-
model: record.model,
|
|
344
349
|
});
|
|
345
350
|
|
|
346
351
|
return new Response(null, { status: 204 });
|
|
@@ -399,7 +404,11 @@ function verifySignature({
|
|
|
399
404
|
|
|
400
405
|
The signature header has the form `t=<unix seconds>,v1=<hex digest>`, where `v1` is the HMAC-SHA256 digest of `"<t>.<raw body>"`. Verify a size-limited raw request body rather than a re-serialized object, use a timing-safe comparison, reject stale timestamps, and reject events whose `data.jobId` does not match the persisted `batch.id`.
|
|
401
406
|
|
|
402
|
-
AI Gateway expects a 2xx response within 10 seconds and retries failed deliveries. It does not follow redirects, so the callback endpoint itself must return the 2xx response. Retries carry the same `x-ai-gateway-idempotency-key` value (`<jobId>-<status>`), which receivers can use for deduplication. After verification, pass the persisted
|
|
407
|
+
AI Gateway expects a 2xx response within 10 seconds and retries failed deliveries. It does not follow redirects, so the callback endpoint itself must return the 2xx response. Retries carry the same `x-ai-gateway-idempotency-key` value (`<jobId>-<status>`), which receivers can use for deduplication. After verification, pass the persisted `batch` reference to `experimental_getBatchStatus` and `experimental_getBatchResults`. Delivery is best-effort, so periodically reconcile persisted batches by status as a fallback for a notification that exhausts its retries.
|
|
408
|
+
|
|
409
|
+
Each request specifies its `type` and `model`. AI Gateway currently requires
|
|
410
|
+
every text request in a batch to use the same model and throws before submission
|
|
411
|
+
when the models differ.
|
|
403
412
|
|
|
404
413
|
## Reranking Models
|
|
405
414
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ai-sdk/gateway",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "4.0.
|
|
4
|
+
"version": "4.0.76",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "Apache-2.0",
|
|
7
7
|
"sideEffects": false,
|
|
@@ -30,8 +30,8 @@
|
|
|
30
30
|
}
|
|
31
31
|
},
|
|
32
32
|
"dependencies": {
|
|
33
|
-
"@ai-sdk/provider": "4.0.
|
|
34
|
-
"@ai-sdk/provider-utils": "5.0.
|
|
33
|
+
"@ai-sdk/provider": "4.0.11",
|
|
34
|
+
"@ai-sdk/provider-utils": "5.0.37",
|
|
35
35
|
"@vercel/oidc": "3.2.0"
|
|
36
36
|
},
|
|
37
37
|
"devDependencies": {
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import {
|
|
2
|
-
|
|
2
|
+
InvalidArgumentError,
|
|
3
|
+
type Experimental_BatchV4 as BatchV4,
|
|
3
4
|
type Experimental_BatchV4ItemResult as BatchV4ItemResult,
|
|
4
5
|
type Experimental_BatchV4OperationOptions as BatchV4OperationOptions,
|
|
5
|
-
type Experimental_BatchV4StartOptions as BatchV4StartOptions,
|
|
6
6
|
type Experimental_BatchV4StartResult as BatchV4StartResult,
|
|
7
7
|
type Experimental_BatchV4Status as BatchV4Status,
|
|
8
|
-
type
|
|
9
|
-
type
|
|
8
|
+
type Experimental_BatchV4StartOptions as BatchV4StartOptions,
|
|
9
|
+
type LanguageModelV4CallOptions,
|
|
10
10
|
type SharedV4ProviderMetadata,
|
|
11
11
|
type SharedV4ProviderOptions,
|
|
12
12
|
} from '@ai-sdk/provider';
|
|
@@ -20,35 +20,20 @@ import {
|
|
|
20
20
|
normalizeBatchRequestCounts,
|
|
21
21
|
postJsonToApi,
|
|
22
22
|
resolve,
|
|
23
|
-
WORKFLOW_SERIALIZE,
|
|
24
|
-
WORKFLOW_DESERIALIZE,
|
|
25
23
|
} from '@ai-sdk/provider-utils';
|
|
26
24
|
import { z } from './zod';
|
|
27
|
-
import {
|
|
28
|
-
GatewayLanguageModel,
|
|
29
|
-
type GatewayChatConfig,
|
|
30
|
-
} from './gateway-language-model';
|
|
25
|
+
import type { GatewayChatConfig } from './gateway-language-model';
|
|
31
26
|
import type { GatewayModelId } from './gateway-language-model-settings';
|
|
32
27
|
import { asGatewayError } from './errors';
|
|
33
28
|
import { parseAuthMethod } from './errors/parse-auth-method';
|
|
34
29
|
|
|
35
|
-
export class
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
{
|
|
39
|
-
static [WORKFLOW_SERIALIZE](model: GatewayBatchLanguageModel) {
|
|
40
|
-
return GatewayLanguageModel[WORKFLOW_SERIALIZE](model);
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
static [WORKFLOW_DESERIALIZE](options: {
|
|
44
|
-
modelId: GatewayModelId;
|
|
45
|
-
config: GatewayChatConfig;
|
|
46
|
-
}) {
|
|
47
|
-
return new GatewayBatchLanguageModel(options.modelId, options.config);
|
|
48
|
-
}
|
|
30
|
+
export class GatewayBatch implements BatchV4<{ text: GatewayModelId }> {
|
|
31
|
+
readonly specificationVersion = 'v4' as const;
|
|
32
|
+
readonly provider: string;
|
|
33
|
+
readonly supportedUrls = { '*/*': [/.*/] };
|
|
49
34
|
|
|
50
|
-
constructor(
|
|
51
|
-
|
|
35
|
+
constructor(private readonly config: GatewayChatConfig) {
|
|
36
|
+
this.provider = `${config.provider}.batch`;
|
|
52
37
|
}
|
|
53
38
|
|
|
54
39
|
/**
|
|
@@ -58,13 +43,17 @@ export class GatewayBatchLanguageModel
|
|
|
58
43
|
* server-side, so status and results always route back through the
|
|
59
44
|
* Gateway job.
|
|
60
45
|
*/
|
|
61
|
-
async
|
|
46
|
+
async doStartBatch({
|
|
62
47
|
requests,
|
|
63
48
|
providerOptions,
|
|
64
49
|
headers,
|
|
65
50
|
abortSignal,
|
|
66
51
|
webhookUrl,
|
|
67
|
-
}: BatchV4StartOptions<
|
|
52
|
+
}: BatchV4StartOptions<{
|
|
53
|
+
text: GatewayModelId;
|
|
54
|
+
}>): Promise<BatchV4StartResult> {
|
|
55
|
+
const modelId = validateSingleModel(requests);
|
|
56
|
+
|
|
68
57
|
const resolvedHeaders = this.config.headers
|
|
69
58
|
? await resolve(this.config.headers)
|
|
70
59
|
: undefined;
|
|
@@ -78,7 +67,7 @@ export class GatewayBatchLanguageModel
|
|
|
78
67
|
headers: combineHeaders(
|
|
79
68
|
resolvedHeaders,
|
|
80
69
|
headers,
|
|
81
|
-
|
|
70
|
+
{ 'ai-model-id': modelId },
|
|
82
71
|
await resolve(this.config.o11yHeaders),
|
|
83
72
|
idempotencyKey != null
|
|
84
73
|
? { 'idempotency-key': idempotencyKey }
|
|
@@ -86,10 +75,11 @@ export class GatewayBatchLanguageModel
|
|
|
86
75
|
),
|
|
87
76
|
body: {
|
|
88
77
|
...(webhookUrl != null && { callbackUrl: webhookUrl }),
|
|
89
|
-
modelId: this.modelId,
|
|
90
78
|
requests: requests.map(request => ({
|
|
91
79
|
id: request.id,
|
|
92
|
-
|
|
80
|
+
type: request.type,
|
|
81
|
+
modelId: request.modelId,
|
|
82
|
+
options: maybeEncodeBatchFileParts(request.options),
|
|
93
83
|
})),
|
|
94
84
|
...(forwardedProviderOptions != null && {
|
|
95
85
|
providerOptions: forwardedProviderOptions,
|
|
@@ -129,7 +119,7 @@ export class GatewayBatchLanguageModel
|
|
|
129
119
|
* Retrieves the lifecycle status of a Gateway batch job
|
|
130
120
|
* (`POST {baseURL}/batch/status`).
|
|
131
121
|
*/
|
|
132
|
-
async
|
|
122
|
+
async doGetBatchStatus({
|
|
133
123
|
batchId,
|
|
134
124
|
headers,
|
|
135
125
|
abortSignal,
|
|
@@ -144,7 +134,6 @@ export class GatewayBatchLanguageModel
|
|
|
144
134
|
headers: combineHeaders(
|
|
145
135
|
resolvedHeaders,
|
|
146
136
|
headers,
|
|
147
|
-
this.getBatchConfigHeaders(),
|
|
148
137
|
await resolve(this.config.o11yHeaders),
|
|
149
138
|
),
|
|
150
139
|
body: { batchId },
|
|
@@ -178,13 +167,11 @@ export class GatewayBatchLanguageModel
|
|
|
178
167
|
* (id + status) and passed through — the Gateway sanitizes them
|
|
179
168
|
* server-side. The route responds 400 while the batch is non-terminal.
|
|
180
169
|
*/
|
|
181
|
-
async
|
|
170
|
+
async doGetBatchResults({
|
|
182
171
|
batchId,
|
|
183
172
|
headers,
|
|
184
173
|
abortSignal,
|
|
185
|
-
}: BatchV4OperationOptions): Promise<
|
|
186
|
-
ReadableStream<BatchV4ItemResult<LanguageModelV4GenerateResult>>
|
|
187
|
-
> {
|
|
174
|
+
}: BatchV4OperationOptions): Promise<ReadableStream<BatchV4ItemResult>> {
|
|
188
175
|
const resolvedHeaders = this.config.headers
|
|
189
176
|
? await resolve(this.config.headers)
|
|
190
177
|
: undefined;
|
|
@@ -195,7 +182,6 @@ export class GatewayBatchLanguageModel
|
|
|
195
182
|
headers: combineHeaders(
|
|
196
183
|
resolvedHeaders,
|
|
197
184
|
headers,
|
|
198
|
-
this.getBatchConfigHeaders(),
|
|
199
185
|
await resolve(this.config.o11yHeaders),
|
|
200
186
|
),
|
|
201
187
|
body: { batchId },
|
|
@@ -227,12 +213,67 @@ export class GatewayBatchLanguageModel
|
|
|
227
213
|
private getBatchUrl(path: 'results' | 'start' | 'status') {
|
|
228
214
|
return `${this.config.baseURL}/batch/${path}`;
|
|
229
215
|
}
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
function maybeEncodeBatchFileParts<
|
|
219
|
+
T extends Pick<LanguageModelV4CallOptions, 'prompt'>,
|
|
220
|
+
>(options: T): T {
|
|
221
|
+
for (const message of options.prompt) {
|
|
222
|
+
if (!Array.isArray(message.content)) {
|
|
223
|
+
continue;
|
|
224
|
+
}
|
|
225
|
+
for (const part of message.content) {
|
|
226
|
+
if (part.type === 'file' || part.type === 'reasoning-file') {
|
|
227
|
+
part.data = maybeBase64EncodeFileData(part.data);
|
|
228
|
+
} else if (
|
|
229
|
+
part.type === 'tool-result' &&
|
|
230
|
+
part.output.type === 'content'
|
|
231
|
+
) {
|
|
232
|
+
for (const contentPart of part.output.value) {
|
|
233
|
+
if (contentPart.type === 'file') {
|
|
234
|
+
contentPart.data = maybeBase64EncodeFileData(contentPart.data);
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
return options;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
function maybeBase64EncodeFileData<T extends { type: string }>(data: T): T {
|
|
244
|
+
if (data.type === 'data') {
|
|
245
|
+
const bytes = (data as { data?: unknown }).data;
|
|
246
|
+
if (bytes instanceof Uint8Array) {
|
|
247
|
+
return { ...data, data: Buffer.from(bytes).toString('base64') } as T;
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
return data;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
function validateSingleModel(
|
|
254
|
+
requests: BatchV4StartOptions<{ text: GatewayModelId }>['requests'],
|
|
255
|
+
): GatewayModelId {
|
|
256
|
+
const modelId = requests[0]?.modelId;
|
|
257
|
+
|
|
258
|
+
if (modelId == null) {
|
|
259
|
+
throw new InvalidArgumentError({
|
|
260
|
+
argument: 'requests',
|
|
261
|
+
message: 'The AI Gateway Batch API requires at least one request.',
|
|
262
|
+
});
|
|
263
|
+
}
|
|
230
264
|
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
265
|
+
for (const request of requests) {
|
|
266
|
+
if (request.modelId !== modelId) {
|
|
267
|
+
throw new InvalidArgumentError({
|
|
268
|
+
argument: 'requests',
|
|
269
|
+
message:
|
|
270
|
+
'The AI Gateway Batch API requires all requests in a batch to use ' +
|
|
271
|
+
`the same model. Found "${modelId}" and "${request.modelId}".`,
|
|
272
|
+
});
|
|
273
|
+
}
|
|
235
274
|
}
|
|
275
|
+
|
|
276
|
+
return modelId;
|
|
236
277
|
}
|
|
237
278
|
|
|
238
279
|
/**
|
|
@@ -360,9 +401,9 @@ function convertGatewayBatchStatus(body: {
|
|
|
360
401
|
*/
|
|
361
402
|
async function* convertGatewayBatchResultLines(
|
|
362
403
|
lines: AsyncIterable<unknown>,
|
|
363
|
-
): AsyncGenerator<BatchV4ItemResult
|
|
404
|
+
): AsyncGenerator<BatchV4ItemResult> {
|
|
364
405
|
for await (const line of lines) {
|
|
365
|
-
const item = line as BatchV4ItemResult
|
|
406
|
+
const item = line as BatchV4ItemResult;
|
|
366
407
|
|
|
367
408
|
// JSON carries `response.timestamp` as an ISO string; core expects a Date.
|
|
368
409
|
if (item.status === 'succeeded') {
|
|
@@ -378,6 +419,7 @@ async function* convertGatewayBatchResultLines(
|
|
|
378
419
|
|
|
379
420
|
const gatewayBatchItemResultLineSchema = z
|
|
380
421
|
.object({
|
|
422
|
+
type: z.literal('text'),
|
|
381
423
|
id: z.string(),
|
|
382
424
|
status: z.enum(['cancelled', 'expired', 'failed', 'succeeded']),
|
|
383
425
|
})
|
|
@@ -64,6 +64,7 @@ export type GatewayModelId =
|
|
|
64
64
|
| 'deepseek/deepseek-v4-flash-vision-exp'
|
|
65
65
|
| 'deepseek/deepseek-v4-pro'
|
|
66
66
|
| 'deepseek/deepseek-v4-pro-0813'
|
|
67
|
+
| 'deepseek/deepseek-v4.1-flash-beta'
|
|
67
68
|
| 'google/gemini-2.5-flash'
|
|
68
69
|
| 'google/gemini-2.5-flash-image'
|
|
69
70
|
| 'google/gemini-2.5-flash-lite'
|
|
@@ -88,6 +89,8 @@ export type GatewayModelId =
|
|
|
88
89
|
| 'inclusionai/ling-3.0-flash'
|
|
89
90
|
| 'inclusionai/ling-3.0-flash-fin'
|
|
90
91
|
| 'inclusionai/ling-3.0-flash-fin-free'
|
|
92
|
+
| 'inclusionai/ling-3.0-flash-sante'
|
|
93
|
+
| 'inclusionai/ling-3.0-flash-sante-free'
|
|
91
94
|
| 'interfaze/interfaze-beta'
|
|
92
95
|
| 'kwaipilot/kat-coder-air-v2.5'
|
|
93
96
|
| 'kwaipilot/kat-coder-pro-v1'
|
|
@@ -110,10 +113,8 @@ export type GatewayModelId =
|
|
|
110
113
|
| 'minimax/minimax-m2.5'
|
|
111
114
|
| 'minimax/minimax-m2.5-highspeed'
|
|
112
115
|
| 'minimax/minimax-m2.7'
|
|
113
|
-
| 'minimax/minimax-m2.7-free'
|
|
114
116
|
| 'minimax/minimax-m2.7-highspeed'
|
|
115
117
|
| 'minimax/minimax-m3'
|
|
116
|
-
| 'minimax/minimax-m3-free'
|
|
117
118
|
| 'mistral/codestral'
|
|
118
119
|
| 'mistral/devstral-2'
|
|
119
120
|
| 'mistral/devstral-small-2'
|
|
@@ -188,6 +189,7 @@ export type GatewayModelId =
|
|
|
188
189
|
| 'openai/gpt-5.6-terra'
|
|
189
190
|
| 'openai/gpt-5.6-terra-fast'
|
|
190
191
|
| 'openai/gpt-6-astra'
|
|
192
|
+
| 'openai/gpt-6-astra-fast'
|
|
191
193
|
| 'openai/gpt-oss-120b'
|
|
192
194
|
| 'openai/gpt-oss-20b'
|
|
193
195
|
| 'openai/gpt-oss-safeguard-120b'
|
|
@@ -22,7 +22,7 @@ export type GatewayProviderOptions = {
|
|
|
22
22
|
has?: Array<'implicit-caching' | 'vision'>;
|
|
23
23
|
|
|
24
24
|
/**
|
|
25
|
-
* Idempotency key for `
|
|
25
|
+
* Idempotency key for `experimental_startBatch`: retries with the same
|
|
26
26
|
* key replay the original batch instead of creating a duplicate.
|
|
27
27
|
*/
|
|
28
28
|
idempotencyKey?: string;
|
package/src/gateway-provider.ts
CHANGED
|
@@ -31,7 +31,8 @@ import {
|
|
|
31
31
|
type GatewayGenerationInfoParams,
|
|
32
32
|
type GatewayGenerationInfo,
|
|
33
33
|
} from './gateway-generation-info';
|
|
34
|
-
import {
|
|
34
|
+
import { GatewayBatch } from './gateway-batch';
|
|
35
|
+
import { GatewayLanguageModel } from './gateway-language-model';
|
|
35
36
|
import { GatewayEmbeddingModel } from './gateway-embedding-model';
|
|
36
37
|
import { GatewayImageModel } from './gateway-image-model';
|
|
37
38
|
import { GatewayVideoModel } from './gateway-video-model';
|
|
@@ -54,7 +55,7 @@ import { getVercelOidcToken, getVercelRequestId } from './vercel-environment';
|
|
|
54
55
|
import type { GatewayModelId } from './gateway-language-model-settings';
|
|
55
56
|
import type {
|
|
56
57
|
EmbeddingModelV4,
|
|
57
|
-
|
|
58
|
+
Experimental_BatchV4 as BatchV4,
|
|
58
59
|
ImageModelV4,
|
|
59
60
|
RerankingModelV4,
|
|
60
61
|
SpeechModelV4,
|
|
@@ -63,21 +64,25 @@ import type {
|
|
|
63
64
|
Experimental_RealtimeFactoryV4 as RealtimeFactoryV4,
|
|
64
65
|
Experimental_RealtimeFactoryV4GetTokenOptions as RealtimeFactoryV4GetTokenOptions,
|
|
65
66
|
ProviderV4,
|
|
67
|
+
LanguageModelV4,
|
|
66
68
|
} from '@ai-sdk/provider';
|
|
67
69
|
import { VERSION } from './version';
|
|
68
70
|
|
|
69
71
|
export interface GatewayProvider extends ProviderV4 {
|
|
70
|
-
(modelId: GatewayModelId):
|
|
72
|
+
(modelId: GatewayModelId): LanguageModelV4;
|
|
71
73
|
|
|
72
74
|
/**
|
|
73
75
|
* Creates a model for text generation.
|
|
74
76
|
*/
|
|
75
|
-
chat(modelId: GatewayModelId):
|
|
77
|
+
chat(modelId: GatewayModelId): LanguageModelV4;
|
|
76
78
|
|
|
77
79
|
/**
|
|
78
80
|
* Creates a model for text generation.
|
|
79
81
|
*/
|
|
80
|
-
languageModel(modelId: GatewayModelId):
|
|
82
|
+
languageModel(modelId: GatewayModelId): LanguageModelV4;
|
|
83
|
+
|
|
84
|
+
/** Returns a BatchV4 interface for processing batches with AI Gateway. */
|
|
85
|
+
experimental_batch(): BatchV4<{ text: GatewayModelId }>;
|
|
81
86
|
|
|
82
87
|
/**
|
|
83
88
|
* Returns available providers and models for use with the remote provider.
|
|
@@ -417,7 +422,7 @@ export function createGateway(
|
|
|
417
422
|
};
|
|
418
423
|
|
|
419
424
|
const createLanguageModel = (modelId: GatewayModelId) => {
|
|
420
|
-
return new
|
|
425
|
+
return new GatewayLanguageModel(modelId, {
|
|
421
426
|
provider: 'gateway',
|
|
422
427
|
baseURL,
|
|
423
428
|
headers: getHeaders,
|
|
@@ -426,6 +431,15 @@ export function createGateway(
|
|
|
426
431
|
});
|
|
427
432
|
};
|
|
428
433
|
|
|
434
|
+
const createBatch = () =>
|
|
435
|
+
new GatewayBatch({
|
|
436
|
+
provider: 'gateway',
|
|
437
|
+
baseURL,
|
|
438
|
+
headers: getHeaders,
|
|
439
|
+
fetch: options.fetch,
|
|
440
|
+
o11yHeaders: createO11yHeaders(),
|
|
441
|
+
});
|
|
442
|
+
|
|
429
443
|
const getAvailableModels = async () => {
|
|
430
444
|
const now = options._internal?.currentDate?.().getTime() ?? Date.now();
|
|
431
445
|
if (!pendingMetadata || now - lastFetchTime > cacheRefreshMillis) {
|
|
@@ -522,6 +536,7 @@ export function createGateway(
|
|
|
522
536
|
});
|
|
523
537
|
};
|
|
524
538
|
provider.languageModel = createLanguageModel;
|
|
539
|
+
provider.experimental_batch = createBatch;
|
|
525
540
|
const createEmbeddingModel = (modelId: GatewayEmbeddingModelId) => {
|
|
526
541
|
return new GatewayEmbeddingModel(modelId, {
|
|
527
542
|
provider: 'gateway',
|