@ai-sdk/gateway 4.0.75 → 4.0.77
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/dist/index.d.ts +11 -7
- package/dist/index.js +295 -244
- package/dist/index.js.map +1 -1
- package/docs/00-ai-gateway.mdx +29 -16
- package/package.json +3 -3
- package/src/{gateway-language-model-batch.ts → gateway-batch.ts} +86 -44
- package/src/gateway-image-model-settings.ts +2 -0
- package/src/gateway-image-model.ts +4 -0
- package/src/gateway-language-model-settings.ts +5 -4
- package/src/gateway-provider-options.ts +1 -1
- package/src/gateway-provider.ts +21 -6
package/docs/00-ai-gateway.mdx
CHANGED
|
@@ -213,21 +213,25 @@ started, AI SDK Core exposes it as a
|
|
|
213
213
|
Use its `type`, `code`, `statusCode`, and `isRetryable` metadata to classify
|
|
214
214
|
the failure without inspecting the Gateway's serialized error payload.
|
|
215
215
|
|
|
216
|
-
##
|
|
216
|
+
## Batch
|
|
217
217
|
|
|
218
218
|
<Note type="warning">
|
|
219
|
-
|
|
219
|
+
Batch support is experimental and the API may change in patch releases.
|
|
220
220
|
</Note>
|
|
221
221
|
|
|
222
|
-
AI Gateway supports durable text batches through `
|
|
222
|
+
AI Gateway supports durable text batches through `experimental_startBatch`,
|
|
223
|
+
`experimental_getBatchStatus`, and `experimental_getBatchResults`. See
|
|
224
|
+
[Batch](/docs/ai-sdk-core/batch) for the provider-agnostic workflow,
|
|
225
|
+
and the [AI Gateway batch processing guide](https://vercel.com/docs/ai-gateway/models-and-providers/batch-processing)
|
|
226
|
+
for supported models, limits, and Gateway-specific behavior.
|
|
223
227
|
|
|
224
228
|
Pass a publicly reachable HTTPS `webhookUrl` when starting a batch to receive a terminal notification instead of polling. AI Gateway sends `batch.completed`, `batch.failed`, or `batch.cancelled`. The event contains terminal status information but not batch results. Completion webhooks are an AI Gateway capability; direct Anthropic and OpenAI batch providers return an unsupported warning when this option is provided.
|
|
225
229
|
|
|
226
230
|
```ts filename="start-batch.ts"
|
|
227
231
|
import { randomUUID } from 'node:crypto';
|
|
228
232
|
import {
|
|
229
|
-
|
|
230
|
-
type
|
|
233
|
+
experimental_startBatch as startBatch,
|
|
234
|
+
type Experimental_BatchReference as BatchReference,
|
|
231
235
|
type GatewayProviderMetadata,
|
|
232
236
|
} from 'ai';
|
|
233
237
|
|
|
@@ -240,11 +244,20 @@ const model = 'anthropic/claude-haiku-4.5' as const;
|
|
|
240
244
|
// pending record rather than an unknown callback.
|
|
241
245
|
await reserveBatchWebhook(token);
|
|
242
246
|
|
|
243
|
-
const batch = await
|
|
244
|
-
model,
|
|
247
|
+
const batch = await startBatch({
|
|
245
248
|
requests: [
|
|
246
|
-
{
|
|
247
|
-
|
|
249
|
+
{
|
|
250
|
+
id: 'france',
|
|
251
|
+
type: 'text',
|
|
252
|
+
model,
|
|
253
|
+
prompt: 'What is the capital of France?',
|
|
254
|
+
},
|
|
255
|
+
{
|
|
256
|
+
id: 'germany',
|
|
257
|
+
type: 'text',
|
|
258
|
+
model,
|
|
259
|
+
prompt: 'What is the capital of Germany?',
|
|
260
|
+
},
|
|
248
261
|
],
|
|
249
262
|
providerOptions: {
|
|
250
263
|
gateway: { idempotencyKey: token },
|
|
@@ -264,20 +277,17 @@ if (typeof signingSecret !== 'string') {
|
|
|
264
277
|
// Keep sensitive provider metadata out of the receiver queue and logs.
|
|
265
278
|
const batchReference = {
|
|
266
279
|
version: batch.version,
|
|
267
|
-
type: batch.type,
|
|
268
280
|
id: batch.id,
|
|
269
281
|
provider: batch.provider,
|
|
270
|
-
|
|
271
|
-
} satisfies TextBatchReference;
|
|
282
|
+
} satisfies BatchReference;
|
|
272
283
|
|
|
273
284
|
await completeBatchWebhookReservation(token, {
|
|
274
285
|
batch: batchReference,
|
|
275
|
-
model,
|
|
276
286
|
signingSecret,
|
|
277
287
|
});
|
|
278
288
|
```
|
|
279
289
|
|
|
280
|
-
The signing secret is returned only in the start response. Reserve the callback token before submitting the batch, then persist the secret
|
|
290
|
+
The signing secret is returned only in the start response. Reserve the callback token before submitting the batch, then persist the secret and minimal `batch` reference before the process exits. The stable `providerOptions.gateway.idempotencyKey` makes retrying an ambiguous start safe. Reuse the reserved token when retrying; do not generate a new one. Delete the reservation after a definitive start failure. While the reservation is still pending, the receiver should return a retryable response so an early delivery is sent again after setup completes.
|
|
281
291
|
|
|
282
292
|
The receiver must verify the `x-ai-gateway-signature` header against a bounded raw request body before trusting the event:
|
|
283
293
|
|
|
@@ -340,7 +350,6 @@ export async function POST(request: Request) {
|
|
|
340
350
|
await enqueueBatchResultFetch({
|
|
341
351
|
batch: record.batch,
|
|
342
352
|
deliveryId,
|
|
343
|
-
model: record.model,
|
|
344
353
|
});
|
|
345
354
|
|
|
346
355
|
return new Response(null, { status: 204 });
|
|
@@ -399,7 +408,11 @@ function verifySignature({
|
|
|
399
408
|
|
|
400
409
|
The signature header has the form `t=<unix seconds>,v1=<hex digest>`, where `v1` is the HMAC-SHA256 digest of `"<t>.<raw body>"`. Verify a size-limited raw request body rather than a re-serialized object, use a timing-safe comparison, reject stale timestamps, and reject events whose `data.jobId` does not match the persisted `batch.id`.
|
|
401
410
|
|
|
402
|
-
AI Gateway expects a 2xx response within 10 seconds and retries failed deliveries. It does not follow redirects, so the callback endpoint itself must return the 2xx response. Retries carry the same `x-ai-gateway-idempotency-key` value (`<jobId>-<status>`), which receivers can use for deduplication. After verification, pass the persisted
|
|
411
|
+
AI Gateway expects a 2xx response within 10 seconds and retries failed deliveries. It does not follow redirects, so the callback endpoint itself must return the 2xx response. Retries carry the same `x-ai-gateway-idempotency-key` value (`<jobId>-<status>`), which receivers can use for deduplication. After verification, pass the persisted `batch` reference to `experimental_getBatchStatus` and `experimental_getBatchResults`. Delivery is best-effort, so periodically reconcile persisted batches by status as a fallback for a notification that exhausts its retries.
|
|
412
|
+
|
|
413
|
+
Each request specifies its `type` and `model`. AI Gateway currently requires
|
|
414
|
+
every text request in a batch to use the same model and throws before submission
|
|
415
|
+
when the models differ.
|
|
403
416
|
|
|
404
417
|
## Reranking Models
|
|
405
418
|
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ai-sdk/gateway",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "4.0.
|
|
4
|
+
"version": "4.0.77",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "Apache-2.0",
|
|
7
7
|
"sideEffects": false,
|
|
@@ -30,8 +30,8 @@
|
|
|
30
30
|
}
|
|
31
31
|
},
|
|
32
32
|
"dependencies": {
|
|
33
|
-
"@ai-sdk/provider": "4.0.
|
|
34
|
-
"@ai-sdk/provider-utils": "5.0.
|
|
33
|
+
"@ai-sdk/provider": "4.0.12",
|
|
34
|
+
"@ai-sdk/provider-utils": "5.0.38",
|
|
35
35
|
"@vercel/oidc": "3.2.0"
|
|
36
36
|
},
|
|
37
37
|
"devDependencies": {
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import {
|
|
2
|
-
|
|
2
|
+
InvalidArgumentError,
|
|
3
|
+
type Experimental_BatchV4 as BatchV4,
|
|
3
4
|
type Experimental_BatchV4ItemResult as BatchV4ItemResult,
|
|
4
5
|
type Experimental_BatchV4OperationOptions as BatchV4OperationOptions,
|
|
5
|
-
type Experimental_BatchV4StartOptions as BatchV4StartOptions,
|
|
6
6
|
type Experimental_BatchV4StartResult as BatchV4StartResult,
|
|
7
7
|
type Experimental_BatchV4Status as BatchV4Status,
|
|
8
|
-
type
|
|
9
|
-
type
|
|
8
|
+
type Experimental_BatchV4StartOptions as BatchV4StartOptions,
|
|
9
|
+
type LanguageModelV4CallOptions,
|
|
10
10
|
type SharedV4ProviderMetadata,
|
|
11
11
|
type SharedV4ProviderOptions,
|
|
12
12
|
} from '@ai-sdk/provider';
|
|
@@ -20,35 +20,20 @@ import {
|
|
|
20
20
|
normalizeBatchRequestCounts,
|
|
21
21
|
postJsonToApi,
|
|
22
22
|
resolve,
|
|
23
|
-
WORKFLOW_SERIALIZE,
|
|
24
|
-
WORKFLOW_DESERIALIZE,
|
|
25
23
|
} from '@ai-sdk/provider-utils';
|
|
26
24
|
import { z } from './zod';
|
|
27
|
-
import {
|
|
28
|
-
GatewayLanguageModel,
|
|
29
|
-
type GatewayChatConfig,
|
|
30
|
-
} from './gateway-language-model';
|
|
25
|
+
import type { GatewayChatConfig } from './gateway-language-model';
|
|
31
26
|
import type { GatewayModelId } from './gateway-language-model-settings';
|
|
32
27
|
import { asGatewayError } from './errors';
|
|
33
28
|
import { parseAuthMethod } from './errors/parse-auth-method';
|
|
34
29
|
|
|
35
|
-
export class
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
{
|
|
39
|
-
static [WORKFLOW_SERIALIZE](model: GatewayBatchLanguageModel) {
|
|
40
|
-
return GatewayLanguageModel[WORKFLOW_SERIALIZE](model);
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
static [WORKFLOW_DESERIALIZE](options: {
|
|
44
|
-
modelId: GatewayModelId;
|
|
45
|
-
config: GatewayChatConfig;
|
|
46
|
-
}) {
|
|
47
|
-
return new GatewayBatchLanguageModel(options.modelId, options.config);
|
|
48
|
-
}
|
|
30
|
+
export class GatewayBatch implements BatchV4<{ text: GatewayModelId }> {
|
|
31
|
+
readonly specificationVersion = 'v4' as const;
|
|
32
|
+
readonly provider: string;
|
|
33
|
+
readonly supportedUrls = { '*/*': [/.*/] };
|
|
49
34
|
|
|
50
|
-
constructor(
|
|
51
|
-
|
|
35
|
+
constructor(private readonly config: GatewayChatConfig) {
|
|
36
|
+
this.provider = `${config.provider}.batch`;
|
|
52
37
|
}
|
|
53
38
|
|
|
54
39
|
/**
|
|
@@ -58,13 +43,17 @@ export class GatewayBatchLanguageModel
|
|
|
58
43
|
* server-side, so status and results always route back through the
|
|
59
44
|
* Gateway job.
|
|
60
45
|
*/
|
|
61
|
-
async
|
|
46
|
+
async doStartBatch({
|
|
62
47
|
requests,
|
|
63
48
|
providerOptions,
|
|
64
49
|
headers,
|
|
65
50
|
abortSignal,
|
|
66
51
|
webhookUrl,
|
|
67
|
-
}: BatchV4StartOptions<
|
|
52
|
+
}: BatchV4StartOptions<{
|
|
53
|
+
text: GatewayModelId;
|
|
54
|
+
}>): Promise<BatchV4StartResult> {
|
|
55
|
+
const modelId = validateSingleModel(requests);
|
|
56
|
+
|
|
68
57
|
const resolvedHeaders = this.config.headers
|
|
69
58
|
? await resolve(this.config.headers)
|
|
70
59
|
: undefined;
|
|
@@ -78,7 +67,7 @@ export class GatewayBatchLanguageModel
|
|
|
78
67
|
headers: combineHeaders(
|
|
79
68
|
resolvedHeaders,
|
|
80
69
|
headers,
|
|
81
|
-
|
|
70
|
+
{ 'ai-model-id': modelId },
|
|
82
71
|
await resolve(this.config.o11yHeaders),
|
|
83
72
|
idempotencyKey != null
|
|
84
73
|
? { 'idempotency-key': idempotencyKey }
|
|
@@ -86,10 +75,11 @@ export class GatewayBatchLanguageModel
|
|
|
86
75
|
),
|
|
87
76
|
body: {
|
|
88
77
|
...(webhookUrl != null && { callbackUrl: webhookUrl }),
|
|
89
|
-
modelId: this.modelId,
|
|
90
78
|
requests: requests.map(request => ({
|
|
91
79
|
id: request.id,
|
|
92
|
-
|
|
80
|
+
type: request.type,
|
|
81
|
+
modelId: request.modelId,
|
|
82
|
+
options: maybeEncodeBatchFileParts(request.options),
|
|
93
83
|
})),
|
|
94
84
|
...(forwardedProviderOptions != null && {
|
|
95
85
|
providerOptions: forwardedProviderOptions,
|
|
@@ -129,7 +119,7 @@ export class GatewayBatchLanguageModel
|
|
|
129
119
|
* Retrieves the lifecycle status of a Gateway batch job
|
|
130
120
|
* (`POST {baseURL}/batch/status`).
|
|
131
121
|
*/
|
|
132
|
-
async
|
|
122
|
+
async doGetBatchStatus({
|
|
133
123
|
batchId,
|
|
134
124
|
headers,
|
|
135
125
|
abortSignal,
|
|
@@ -144,7 +134,6 @@ export class GatewayBatchLanguageModel
|
|
|
144
134
|
headers: combineHeaders(
|
|
145
135
|
resolvedHeaders,
|
|
146
136
|
headers,
|
|
147
|
-
this.getBatchConfigHeaders(),
|
|
148
137
|
await resolve(this.config.o11yHeaders),
|
|
149
138
|
),
|
|
150
139
|
body: { batchId },
|
|
@@ -178,13 +167,11 @@ export class GatewayBatchLanguageModel
|
|
|
178
167
|
* (id + status) and passed through — the Gateway sanitizes them
|
|
179
168
|
* server-side. The route responds 400 while the batch is non-terminal.
|
|
180
169
|
*/
|
|
181
|
-
async
|
|
170
|
+
async doGetBatchResults({
|
|
182
171
|
batchId,
|
|
183
172
|
headers,
|
|
184
173
|
abortSignal,
|
|
185
|
-
}: BatchV4OperationOptions): Promise<
|
|
186
|
-
ReadableStream<BatchV4ItemResult<LanguageModelV4GenerateResult>>
|
|
187
|
-
> {
|
|
174
|
+
}: BatchV4OperationOptions): Promise<ReadableStream<BatchV4ItemResult>> {
|
|
188
175
|
const resolvedHeaders = this.config.headers
|
|
189
176
|
? await resolve(this.config.headers)
|
|
190
177
|
: undefined;
|
|
@@ -195,7 +182,6 @@ export class GatewayBatchLanguageModel
|
|
|
195
182
|
headers: combineHeaders(
|
|
196
183
|
resolvedHeaders,
|
|
197
184
|
headers,
|
|
198
|
-
this.getBatchConfigHeaders(),
|
|
199
185
|
await resolve(this.config.o11yHeaders),
|
|
200
186
|
),
|
|
201
187
|
body: { batchId },
|
|
@@ -227,12 +213,67 @@ export class GatewayBatchLanguageModel
|
|
|
227
213
|
private getBatchUrl(path: 'results' | 'start' | 'status') {
|
|
228
214
|
return `${this.config.baseURL}/batch/${path}`;
|
|
229
215
|
}
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
function maybeEncodeBatchFileParts<
|
|
219
|
+
T extends Pick<LanguageModelV4CallOptions, 'prompt'>,
|
|
220
|
+
>(options: T): T {
|
|
221
|
+
for (const message of options.prompt) {
|
|
222
|
+
if (!Array.isArray(message.content)) {
|
|
223
|
+
continue;
|
|
224
|
+
}
|
|
225
|
+
for (const part of message.content) {
|
|
226
|
+
if (part.type === 'file' || part.type === 'reasoning-file') {
|
|
227
|
+
part.data = maybeBase64EncodeFileData(part.data);
|
|
228
|
+
} else if (
|
|
229
|
+
part.type === 'tool-result' &&
|
|
230
|
+
part.output.type === 'content'
|
|
231
|
+
) {
|
|
232
|
+
for (const contentPart of part.output.value) {
|
|
233
|
+
if (contentPart.type === 'file') {
|
|
234
|
+
contentPart.data = maybeBase64EncodeFileData(contentPart.data);
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
return options;
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
function maybeBase64EncodeFileData<T extends { type: string }>(data: T): T {
|
|
244
|
+
if (data.type === 'data') {
|
|
245
|
+
const bytes = (data as { data?: unknown }).data;
|
|
246
|
+
if (bytes instanceof Uint8Array) {
|
|
247
|
+
return { ...data, data: Buffer.from(bytes).toString('base64') } as T;
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
return data;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
function validateSingleModel(
|
|
254
|
+
requests: BatchV4StartOptions<{ text: GatewayModelId }>['requests'],
|
|
255
|
+
): GatewayModelId {
|
|
256
|
+
const modelId = requests[0]?.modelId;
|
|
257
|
+
|
|
258
|
+
if (modelId == null) {
|
|
259
|
+
throw new InvalidArgumentError({
|
|
260
|
+
argument: 'requests',
|
|
261
|
+
message: 'The AI Gateway Batch API requires at least one request.',
|
|
262
|
+
});
|
|
263
|
+
}
|
|
230
264
|
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
265
|
+
for (const request of requests) {
|
|
266
|
+
if (request.modelId !== modelId) {
|
|
267
|
+
throw new InvalidArgumentError({
|
|
268
|
+
argument: 'requests',
|
|
269
|
+
message:
|
|
270
|
+
'The AI Gateway Batch API requires all requests in a batch to use ' +
|
|
271
|
+
`the same model. Found "${modelId}" and "${request.modelId}".`,
|
|
272
|
+
});
|
|
273
|
+
}
|
|
235
274
|
}
|
|
275
|
+
|
|
276
|
+
return modelId;
|
|
236
277
|
}
|
|
237
278
|
|
|
238
279
|
/**
|
|
@@ -360,9 +401,9 @@ function convertGatewayBatchStatus(body: {
|
|
|
360
401
|
*/
|
|
361
402
|
async function* convertGatewayBatchResultLines(
|
|
362
403
|
lines: AsyncIterable<unknown>,
|
|
363
|
-
): AsyncGenerator<BatchV4ItemResult
|
|
404
|
+
): AsyncGenerator<BatchV4ItemResult> {
|
|
364
405
|
for await (const line of lines) {
|
|
365
|
-
const item = line as BatchV4ItemResult
|
|
406
|
+
const item = line as BatchV4ItemResult;
|
|
366
407
|
|
|
367
408
|
// JSON carries `response.timestamp` as an ISO string; core expects a Date.
|
|
368
409
|
if (item.status === 'succeeded') {
|
|
@@ -378,6 +419,7 @@ async function* convertGatewayBatchResultLines(
|
|
|
378
419
|
|
|
379
420
|
const gatewayBatchItemResultLineSchema = z
|
|
380
421
|
.object({
|
|
422
|
+
type: z.literal('text'),
|
|
381
423
|
id: z.string(),
|
|
382
424
|
status: z.enum(['cancelled', 'expired', 'failed', 'succeeded']),
|
|
383
425
|
})
|
|
@@ -18,6 +18,8 @@ export type GatewayImageModelId =
|
|
|
18
18
|
| 'openai/gpt-image-1-mini'
|
|
19
19
|
| 'openai/gpt-image-1.5'
|
|
20
20
|
| 'openai/gpt-image-2'
|
|
21
|
+
| 'openai/gpt-image-2.5-flare'
|
|
22
|
+
| 'openai/gpt-image-2.5-sunburst'
|
|
21
23
|
| 'prodia/flux-fast-schnell'
|
|
22
24
|
| 'quiverai/arrow-1.1'
|
|
23
25
|
| 'recraft/recraft-v2'
|
|
@@ -105,6 +105,9 @@ export class GatewayImageModel implements ImageModelV4 {
|
|
|
105
105
|
|
|
106
106
|
return {
|
|
107
107
|
images: responseBody.images, // Always base64 strings from server
|
|
108
|
+
...(responseBody.isRetryable != null && {
|
|
109
|
+
isRetryable: responseBody.isRetryable,
|
|
110
|
+
}),
|
|
108
111
|
warnings: responseBody.warnings ?? [],
|
|
109
112
|
providerMetadata:
|
|
110
113
|
responseBody.providerMetadata as ImageModelV4ProviderMetadata,
|
|
@@ -187,6 +190,7 @@ const gatewayImageUsageSchema = z.object({
|
|
|
187
190
|
|
|
188
191
|
const gatewayImageResponseSchema = z.object({
|
|
189
192
|
images: z.array(z.string()), // Always base64 strings over the wire
|
|
193
|
+
isRetryable: z.boolean().optional(),
|
|
190
194
|
warnings: z.array(gatewayImageWarningSchema).optional(),
|
|
191
195
|
providerMetadata: z
|
|
192
196
|
.record(z.string(), providerMetadataEntrySchema)
|
|
@@ -64,6 +64,7 @@ export type GatewayModelId =
|
|
|
64
64
|
| 'deepseek/deepseek-v4-flash-vision-exp'
|
|
65
65
|
| 'deepseek/deepseek-v4-pro'
|
|
66
66
|
| 'deepseek/deepseek-v4-pro-0813'
|
|
67
|
+
| 'deepseek/deepseek-v4.1-flash-beta'
|
|
67
68
|
| 'google/gemini-2.5-flash'
|
|
68
69
|
| 'google/gemini-2.5-flash-image'
|
|
69
70
|
| 'google/gemini-2.5-flash-lite'
|
|
@@ -84,10 +85,13 @@ export type GatewayModelId =
|
|
|
84
85
|
| 'google/gemma-4-26b-a4b-it'
|
|
85
86
|
| 'google/gemma-4-31b-it'
|
|
86
87
|
| 'inception/mercury-2'
|
|
88
|
+
| 'inception/mercury-2.5'
|
|
87
89
|
| 'inception/mercury-coder-small'
|
|
88
90
|
| 'inclusionai/ling-3.0-flash'
|
|
89
91
|
| 'inclusionai/ling-3.0-flash-fin'
|
|
90
92
|
| 'inclusionai/ling-3.0-flash-fin-free'
|
|
93
|
+
| 'inclusionai/ling-3.0-flash-sante'
|
|
94
|
+
| 'inclusionai/ling-3.0-flash-sante-free'
|
|
91
95
|
| 'interfaze/interfaze-beta'
|
|
92
96
|
| 'kwaipilot/kat-coder-air-v2.5'
|
|
93
97
|
| 'kwaipilot/kat-coder-pro-v1'
|
|
@@ -110,10 +114,8 @@ export type GatewayModelId =
|
|
|
110
114
|
| 'minimax/minimax-m2.5'
|
|
111
115
|
| 'minimax/minimax-m2.5-highspeed'
|
|
112
116
|
| 'minimax/minimax-m2.7'
|
|
113
|
-
| 'minimax/minimax-m2.7-free'
|
|
114
117
|
| 'minimax/minimax-m2.7-highspeed'
|
|
115
118
|
| 'minimax/minimax-m3'
|
|
116
|
-
| 'minimax/minimax-m3-free'
|
|
117
119
|
| 'mistral/codestral'
|
|
118
120
|
| 'mistral/devstral-2'
|
|
119
121
|
| 'mistral/devstral-small-2'
|
|
@@ -188,6 +190,7 @@ export type GatewayModelId =
|
|
|
188
190
|
| 'openai/gpt-5.6-terra'
|
|
189
191
|
| 'openai/gpt-5.6-terra-fast'
|
|
190
192
|
| 'openai/gpt-6-astra'
|
|
193
|
+
| 'openai/gpt-6-astra-fast'
|
|
191
194
|
| 'openai/gpt-oss-120b'
|
|
192
195
|
| 'openai/gpt-oss-20b'
|
|
193
196
|
| 'openai/gpt-oss-safeguard-120b'
|
|
@@ -229,7 +232,6 @@ export type GatewayModelId =
|
|
|
229
232
|
| 'thinkingmachines/inkling-small'
|
|
230
233
|
| 'xiaomi/mimo-v2.5'
|
|
231
234
|
| 'xiaomi/mimo-v2.5-pro'
|
|
232
|
-
| 'xiaomi/mimo-v2.5-pro-ultraspeed'
|
|
233
235
|
| 'zai/glm-4.5'
|
|
234
236
|
| 'zai/glm-4.5-air'
|
|
235
237
|
| 'zai/glm-4.5v'
|
|
@@ -245,6 +247,5 @@ export type GatewayModelId =
|
|
|
245
247
|
| 'zai/glm-5.3'
|
|
246
248
|
| 'zai/glm-5.3-fast'
|
|
247
249
|
| 'zai/glm-5.3-flash'
|
|
248
|
-
| 'zai/glm-5.3-promo-50'
|
|
249
250
|
| 'zai/glm-5v-turbo'
|
|
250
251
|
| (string & {});
|
|
@@ -22,7 +22,7 @@ export type GatewayProviderOptions = {
|
|
|
22
22
|
has?: Array<'implicit-caching' | 'vision'>;
|
|
23
23
|
|
|
24
24
|
/**
|
|
25
|
-
* Idempotency key for `
|
|
25
|
+
* Idempotency key for `experimental_startBatch`: retries with the same
|
|
26
26
|
* key replay the original batch instead of creating a duplicate.
|
|
27
27
|
*/
|
|
28
28
|
idempotencyKey?: string;
|
package/src/gateway-provider.ts
CHANGED
|
@@ -31,7 +31,8 @@ import {
|
|
|
31
31
|
type GatewayGenerationInfoParams,
|
|
32
32
|
type GatewayGenerationInfo,
|
|
33
33
|
} from './gateway-generation-info';
|
|
34
|
-
import {
|
|
34
|
+
import { GatewayBatch } from './gateway-batch';
|
|
35
|
+
import { GatewayLanguageModel } from './gateway-language-model';
|
|
35
36
|
import { GatewayEmbeddingModel } from './gateway-embedding-model';
|
|
36
37
|
import { GatewayImageModel } from './gateway-image-model';
|
|
37
38
|
import { GatewayVideoModel } from './gateway-video-model';
|
|
@@ -54,7 +55,7 @@ import { getVercelOidcToken, getVercelRequestId } from './vercel-environment';
|
|
|
54
55
|
import type { GatewayModelId } from './gateway-language-model-settings';
|
|
55
56
|
import type {
|
|
56
57
|
EmbeddingModelV4,
|
|
57
|
-
|
|
58
|
+
Experimental_BatchV4 as BatchV4,
|
|
58
59
|
ImageModelV4,
|
|
59
60
|
RerankingModelV4,
|
|
60
61
|
SpeechModelV4,
|
|
@@ -63,21 +64,25 @@ import type {
|
|
|
63
64
|
Experimental_RealtimeFactoryV4 as RealtimeFactoryV4,
|
|
64
65
|
Experimental_RealtimeFactoryV4GetTokenOptions as RealtimeFactoryV4GetTokenOptions,
|
|
65
66
|
ProviderV4,
|
|
67
|
+
LanguageModelV4,
|
|
66
68
|
} from '@ai-sdk/provider';
|
|
67
69
|
import { VERSION } from './version';
|
|
68
70
|
|
|
69
71
|
export interface GatewayProvider extends ProviderV4 {
|
|
70
|
-
(modelId: GatewayModelId):
|
|
72
|
+
(modelId: GatewayModelId): LanguageModelV4;
|
|
71
73
|
|
|
72
74
|
/**
|
|
73
75
|
* Creates a model for text generation.
|
|
74
76
|
*/
|
|
75
|
-
chat(modelId: GatewayModelId):
|
|
77
|
+
chat(modelId: GatewayModelId): LanguageModelV4;
|
|
76
78
|
|
|
77
79
|
/**
|
|
78
80
|
* Creates a model for text generation.
|
|
79
81
|
*/
|
|
80
|
-
languageModel(modelId: GatewayModelId):
|
|
82
|
+
languageModel(modelId: GatewayModelId): LanguageModelV4;
|
|
83
|
+
|
|
84
|
+
/** Returns a BatchV4 interface for processing batches with AI Gateway. */
|
|
85
|
+
experimental_batch(): BatchV4<{ text: GatewayModelId }>;
|
|
81
86
|
|
|
82
87
|
/**
|
|
83
88
|
* Returns available providers and models for use with the remote provider.
|
|
@@ -417,7 +422,7 @@ export function createGateway(
|
|
|
417
422
|
};
|
|
418
423
|
|
|
419
424
|
const createLanguageModel = (modelId: GatewayModelId) => {
|
|
420
|
-
return new
|
|
425
|
+
return new GatewayLanguageModel(modelId, {
|
|
421
426
|
provider: 'gateway',
|
|
422
427
|
baseURL,
|
|
423
428
|
headers: getHeaders,
|
|
@@ -426,6 +431,15 @@ export function createGateway(
|
|
|
426
431
|
});
|
|
427
432
|
};
|
|
428
433
|
|
|
434
|
+
const createBatch = () =>
|
|
435
|
+
new GatewayBatch({
|
|
436
|
+
provider: 'gateway',
|
|
437
|
+
baseURL,
|
|
438
|
+
headers: getHeaders,
|
|
439
|
+
fetch: options.fetch,
|
|
440
|
+
o11yHeaders: createO11yHeaders(),
|
|
441
|
+
});
|
|
442
|
+
|
|
429
443
|
const getAvailableModels = async () => {
|
|
430
444
|
const now = options._internal?.currentDate?.().getTime() ?? Date.now();
|
|
431
445
|
if (!pendingMetadata || now - lastFetchTime > cacheRefreshMillis) {
|
|
@@ -522,6 +536,7 @@ export function createGateway(
|
|
|
522
536
|
});
|
|
523
537
|
};
|
|
524
538
|
provider.languageModel = createLanguageModel;
|
|
539
|
+
provider.experimental_batch = createBatch;
|
|
525
540
|
const createEmbeddingModel = (modelId: GatewayEmbeddingModelId) => {
|
|
526
541
|
return new GatewayEmbeddingModel(modelId, {
|
|
527
542
|
provider: 'gateway',
|