@ai-sdk/gateway 4.0.74 → 4.0.76

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -219,15 +219,15 @@ the failure without inspecting the Gateway's serialized error payload.
219
219
  Text batch support is experimental and the API may change in patch releases.
220
220
  </Note>
221
221
 
222
- AI Gateway supports durable text batches through `experimental_startTextBatch`, `experimental_getBatchStatus`, and `experimental_getBatchResults`. See the [AI Gateway batch processing guide](https://vercel.com/docs/ai-gateway/models-and-providers/batch-processing) for supported models, limits, and the complete lifecycle.
222
+ AI Gateway supports durable text batches through `experimental_startBatch`, `experimental_getBatchStatus`, and `experimental_getBatchResults`. See the [AI Gateway batch processing guide](https://vercel.com/docs/ai-gateway/models-and-providers/batch-processing) for supported models, limits, and the complete lifecycle.
223
223
 
224
224
  Pass a publicly reachable HTTPS `webhookUrl` when starting a batch to receive a terminal notification instead of polling. AI Gateway sends `batch.completed`, `batch.failed`, or `batch.cancelled`. The event contains terminal status information but not batch results. Completion webhooks are an AI Gateway capability; direct Anthropic and OpenAI batch providers return an unsupported warning when this option is provided.
225
225
 
226
226
  ```ts filename="start-batch.ts"
227
227
  import { randomUUID } from 'node:crypto';
228
228
  import {
229
- experimental_startTextBatch as startTextBatch,
230
- type Experimental_TextBatchReference as TextBatchReference,
229
+ experimental_startBatch as startBatch,
230
+ type Experimental_BatchReference as BatchReference,
231
231
  type GatewayProviderMetadata,
232
232
  } from 'ai';
233
233
 
@@ -240,11 +240,20 @@ const model = 'anthropic/claude-haiku-4.5' as const;
240
240
  // pending record rather than an unknown callback.
241
241
  await reserveBatchWebhook(token);
242
242
 
243
- const batch = await startTextBatch({
244
- model,
243
+ const batch = await startBatch({
245
244
  requests: [
246
- { id: 'france', prompt: 'What is the capital of France?' },
247
- { id: 'germany', prompt: 'What is the capital of Germany?' },
245
+ {
246
+ id: 'france',
247
+ type: 'text',
248
+ model,
249
+ prompt: 'What is the capital of France?',
250
+ },
251
+ {
252
+ id: 'germany',
253
+ type: 'text',
254
+ model,
255
+ prompt: 'What is the capital of Germany?',
256
+ },
248
257
  ],
249
258
  providerOptions: {
250
259
  gateway: { idempotencyKey: token },
@@ -264,20 +273,17 @@ if (typeof signingSecret !== 'string') {
264
273
  // Keep sensitive provider metadata out of the receiver queue and logs.
265
274
  const batchReference = {
266
275
  version: batch.version,
267
- type: batch.type,
268
276
  id: batch.id,
269
277
  provider: batch.provider,
270
- modelId: batch.modelId,
271
- } satisfies TextBatchReference;
278
+ } satisfies BatchReference;
272
279
 
273
280
  await completeBatchWebhookReservation(token, {
274
281
  batch: batchReference,
275
- model,
276
282
  signingSecret,
277
283
  });
278
284
  ```
279
285
 
280
- The signing secret is returned only in the start response. Reserve the callback token before submitting the batch, then persist the secret, model, and minimal `batch` reference before the process exits. The stable `providerOptions.gateway.idempotencyKey` makes retrying an ambiguous start safe. Reuse the reserved token when retrying; do not generate a new one. Delete the reservation after a definitive start failure. While the reservation is still pending, the receiver should return a retryable response so an early delivery is sent again after setup completes.
286
+ The signing secret is returned only in the start response. Reserve the callback token before submitting the batch, then persist the secret and minimal `batch` reference before the process exits. The stable `providerOptions.gateway.idempotencyKey` makes retrying an ambiguous start safe. Reuse the reserved token when retrying; do not generate a new one. Delete the reservation after a definitive start failure. While the reservation is still pending, the receiver should return a retryable response so an early delivery is sent again after setup completes.
281
287
 
282
288
  The receiver must verify the `x-ai-gateway-signature` header against a bounded raw request body before trusting the event:
283
289
 
@@ -340,7 +346,6 @@ export async function POST(request: Request) {
340
346
  await enqueueBatchResultFetch({
341
347
  batch: record.batch,
342
348
  deliveryId,
343
- model: record.model,
344
349
  });
345
350
 
346
351
  return new Response(null, { status: 204 });
@@ -399,7 +404,11 @@ function verifySignature({
399
404
 
400
405
  The signature header has the form `t=<unix seconds>,v1=<hex digest>`, where `v1` is the HMAC-SHA256 digest of `"<t>.<raw body>"`. Verify a size-limited raw request body rather than a re-serialized object, use a timing-safe comparison, reject stale timestamps, and reject events whose `data.jobId` does not match the persisted `batch.id`.
401
406
 
402
- AI Gateway expects a 2xx response within 10 seconds and retries failed deliveries. It does not follow redirects, so the callback endpoint itself must return the 2xx response. Retries carry the same `x-ai-gateway-idempotency-key` value (`<jobId>-<status>`), which receivers can use for deduplication. After verification, pass the persisted model and `batch` reference to `experimental_getBatchStatus` and `experimental_getBatchResults`. Delivery is best-effort, so periodically reconcile persisted batches by status as a fallback for a notification that exhausts its retries.
407
+ AI Gateway expects a 2xx response within 10 seconds and retries failed deliveries. It does not follow redirects, so the callback endpoint itself must return the 2xx response. Retries carry the same `x-ai-gateway-idempotency-key` value (`<jobId>-<status>`), which receivers can use for deduplication. After verification, pass the persisted `batch` reference to `experimental_getBatchStatus` and `experimental_getBatchResults`. Delivery is best-effort, so periodically reconcile persisted batches by status as a fallback for a notification that exhausts its retries.
408
+
409
+ Each request specifies its `type` and `model`. AI Gateway currently requires
410
+ every text request in a batch to use the same model and throws before submission
411
+ when the models differ.
403
412
 
404
413
  ## Reranking Models
405
414
 
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@ai-sdk/gateway",
3
3
  "private": false,
4
- "version": "4.0.74",
4
+ "version": "4.0.76",
5
5
  "type": "module",
6
6
  "license": "Apache-2.0",
7
7
  "sideEffects": false,
@@ -30,8 +30,8 @@
30
30
  }
31
31
  },
32
32
  "dependencies": {
33
- "@ai-sdk/provider": "4.0.10",
34
- "@ai-sdk/provider-utils": "5.0.36",
33
+ "@ai-sdk/provider": "4.0.11",
34
+ "@ai-sdk/provider-utils": "5.0.37",
35
35
  "@vercel/oidc": "3.2.0"
36
36
  },
37
37
  "devDependencies": {
@@ -1,12 +1,12 @@
1
1
  import {
2
- type Experimental_BatchLanguageModelV4 as BatchLanguageModelV4,
2
+ InvalidArgumentError,
3
+ type Experimental_BatchV4 as BatchV4,
3
4
  type Experimental_BatchV4ItemResult as BatchV4ItemResult,
4
5
  type Experimental_BatchV4OperationOptions as BatchV4OperationOptions,
5
- type Experimental_BatchV4StartOptions as BatchV4StartOptions,
6
6
  type Experimental_BatchV4StartResult as BatchV4StartResult,
7
7
  type Experimental_BatchV4Status as BatchV4Status,
8
- type Experimental_LanguageModelV4BatchRequest as LanguageModelV4BatchRequest,
9
- type LanguageModelV4GenerateResult,
8
+ type Experimental_BatchV4StartOptions as BatchV4StartOptions,
9
+ type LanguageModelV4CallOptions,
10
10
  type SharedV4ProviderMetadata,
11
11
  type SharedV4ProviderOptions,
12
12
  } from '@ai-sdk/provider';
@@ -20,35 +20,20 @@ import {
20
20
  normalizeBatchRequestCounts,
21
21
  postJsonToApi,
22
22
  resolve,
23
- WORKFLOW_SERIALIZE,
24
- WORKFLOW_DESERIALIZE,
25
23
  } from '@ai-sdk/provider-utils';
26
24
  import { z } from './zod';
27
- import {
28
- GatewayLanguageModel,
29
- type GatewayChatConfig,
30
- } from './gateway-language-model';
25
+ import type { GatewayChatConfig } from './gateway-language-model';
31
26
  import type { GatewayModelId } from './gateway-language-model-settings';
32
27
  import { asGatewayError } from './errors';
33
28
  import { parseAuthMethod } from './errors/parse-auth-method';
34
29
 
35
- export class GatewayBatchLanguageModel
36
- extends GatewayLanguageModel
37
- implements BatchLanguageModelV4
38
- {
39
- static [WORKFLOW_SERIALIZE](model: GatewayBatchLanguageModel) {
40
- return GatewayLanguageModel[WORKFLOW_SERIALIZE](model);
41
- }
42
-
43
- static [WORKFLOW_DESERIALIZE](options: {
44
- modelId: GatewayModelId;
45
- config: GatewayChatConfig;
46
- }) {
47
- return new GatewayBatchLanguageModel(options.modelId, options.config);
48
- }
30
+ export class GatewayBatch implements BatchV4<{ text: GatewayModelId }> {
31
+ readonly specificationVersion = 'v4' as const;
32
+ readonly provider: string;
33
+ readonly supportedUrls = { '*/*': [/.*/] };
49
34
 
50
- constructor(modelId: GatewayModelId, config: GatewayChatConfig) {
51
- super(modelId, config);
35
+ constructor(private readonly config: GatewayChatConfig) {
36
+ this.provider = `${config.provider}.batch`;
52
37
  }
53
38
 
54
39
  /**
@@ -58,13 +43,17 @@ export class GatewayBatchLanguageModel
58
43
  * server-side, so status and results always route back through the
59
44
  * Gateway job.
60
45
  */
61
- async experimental_doStartBatch({
46
+ async doStartBatch({
62
47
  requests,
63
48
  providerOptions,
64
49
  headers,
65
50
  abortSignal,
66
51
  webhookUrl,
67
- }: BatchV4StartOptions<LanguageModelV4BatchRequest>): Promise<BatchV4StartResult> {
52
+ }: BatchV4StartOptions<{
53
+ text: GatewayModelId;
54
+ }>): Promise<BatchV4StartResult> {
55
+ const modelId = validateSingleModel(requests);
56
+
68
57
  const resolvedHeaders = this.config.headers
69
58
  ? await resolve(this.config.headers)
70
59
  : undefined;
@@ -78,7 +67,7 @@ export class GatewayBatchLanguageModel
78
67
  headers: combineHeaders(
79
68
  resolvedHeaders,
80
69
  headers,
81
- this.getBatchConfigHeaders(),
70
+ { 'ai-model-id': modelId },
82
71
  await resolve(this.config.o11yHeaders),
83
72
  idempotencyKey != null
84
73
  ? { 'idempotency-key': idempotencyKey }
@@ -86,10 +75,11 @@ export class GatewayBatchLanguageModel
86
75
  ),
87
76
  body: {
88
77
  ...(webhookUrl != null && { callbackUrl: webhookUrl }),
89
- modelId: this.modelId,
90
78
  requests: requests.map(request => ({
91
79
  id: request.id,
92
- options: this.maybeEncodeFileParts(request.options),
80
+ type: request.type,
81
+ modelId: request.modelId,
82
+ options: maybeEncodeBatchFileParts(request.options),
93
83
  })),
94
84
  ...(forwardedProviderOptions != null && {
95
85
  providerOptions: forwardedProviderOptions,
@@ -129,7 +119,7 @@ export class GatewayBatchLanguageModel
129
119
  * Retrieves the lifecycle status of a Gateway batch job
130
120
  * (`POST {baseURL}/batch/status`).
131
121
  */
132
- async experimental_doGetBatchStatus({
122
+ async doGetBatchStatus({
133
123
  batchId,
134
124
  headers,
135
125
  abortSignal,
@@ -144,7 +134,6 @@ export class GatewayBatchLanguageModel
144
134
  headers: combineHeaders(
145
135
  resolvedHeaders,
146
136
  headers,
147
- this.getBatchConfigHeaders(),
148
137
  await resolve(this.config.o11yHeaders),
149
138
  ),
150
139
  body: { batchId },
@@ -178,13 +167,11 @@ export class GatewayBatchLanguageModel
178
167
  * (id + status) and passed through — the Gateway sanitizes them
179
168
  * server-side. The route responds 400 while the batch is non-terminal.
180
169
  */
181
- async experimental_doGetBatchResults({
170
+ async doGetBatchResults({
182
171
  batchId,
183
172
  headers,
184
173
  abortSignal,
185
- }: BatchV4OperationOptions): Promise<
186
- ReadableStream<BatchV4ItemResult<LanguageModelV4GenerateResult>>
187
- > {
174
+ }: BatchV4OperationOptions): Promise<ReadableStream<BatchV4ItemResult>> {
188
175
  const resolvedHeaders = this.config.headers
189
176
  ? await resolve(this.config.headers)
190
177
  : undefined;
@@ -195,7 +182,6 @@ export class GatewayBatchLanguageModel
195
182
  headers: combineHeaders(
196
183
  resolvedHeaders,
197
184
  headers,
198
- this.getBatchConfigHeaders(),
199
185
  await resolve(this.config.o11yHeaders),
200
186
  ),
201
187
  body: { batchId },
@@ -227,12 +213,67 @@ export class GatewayBatchLanguageModel
227
213
  private getBatchUrl(path: 'results' | 'start' | 'status') {
228
214
  return `${this.config.baseURL}/batch/${path}`;
229
215
  }
216
+ }
217
+
218
+ function maybeEncodeBatchFileParts<
219
+ T extends Pick<LanguageModelV4CallOptions, 'prompt'>,
220
+ >(options: T): T {
221
+ for (const message of options.prompt) {
222
+ if (!Array.isArray(message.content)) {
223
+ continue;
224
+ }
225
+ for (const part of message.content) {
226
+ if (part.type === 'file' || part.type === 'reasoning-file') {
227
+ part.data = maybeBase64EncodeFileData(part.data);
228
+ } else if (
229
+ part.type === 'tool-result' &&
230
+ part.output.type === 'content'
231
+ ) {
232
+ for (const contentPart of part.output.value) {
233
+ if (contentPart.type === 'file') {
234
+ contentPart.data = maybeBase64EncodeFileData(contentPart.data);
235
+ }
236
+ }
237
+ }
238
+ }
239
+ }
240
+ return options;
241
+ }
242
+
243
+ function maybeBase64EncodeFileData<T extends { type: string }>(data: T): T {
244
+ if (data.type === 'data') {
245
+ const bytes = (data as { data?: unknown }).data;
246
+ if (bytes instanceof Uint8Array) {
247
+ return { ...data, data: Buffer.from(bytes).toString('base64') } as T;
248
+ }
249
+ }
250
+ return data;
251
+ }
252
+
253
+ function validateSingleModel(
254
+ requests: BatchV4StartOptions<{ text: GatewayModelId }>['requests'],
255
+ ): GatewayModelId {
256
+ const modelId = requests[0]?.modelId;
257
+
258
+ if (modelId == null) {
259
+ throw new InvalidArgumentError({
260
+ argument: 'requests',
261
+ message: 'The AI Gateway Batch API requires at least one request.',
262
+ });
263
+ }
230
264
 
231
- private getBatchConfigHeaders() {
232
- return {
233
- 'ai-model-id': this.modelId,
234
- };
265
+ for (const request of requests) {
266
+ if (request.modelId !== modelId) {
267
+ throw new InvalidArgumentError({
268
+ argument: 'requests',
269
+ message:
270
+ 'The AI Gateway Batch API requires all requests in a batch to use ' +
271
+ `the same model. Found "${modelId}" and "${request.modelId}".`,
272
+ });
273
+ }
235
274
  }
275
+
276
+ return modelId;
236
277
  }
237
278
 
238
279
  /**
@@ -360,9 +401,9 @@ function convertGatewayBatchStatus(body: {
360
401
  */
361
402
  async function* convertGatewayBatchResultLines(
362
403
  lines: AsyncIterable<unknown>,
363
- ): AsyncGenerator<BatchV4ItemResult<LanguageModelV4GenerateResult>> {
404
+ ): AsyncGenerator<BatchV4ItemResult> {
364
405
  for await (const line of lines) {
365
- const item = line as BatchV4ItemResult<LanguageModelV4GenerateResult>;
406
+ const item = line as BatchV4ItemResult;
366
407
 
367
408
  // JSON carries `response.timestamp` as an ISO string; core expects a Date.
368
409
  if (item.status === 'succeeded') {
@@ -378,6 +419,7 @@ async function* convertGatewayBatchResultLines(
378
419
 
379
420
  const gatewayBatchItemResultLineSchema = z
380
421
  .object({
422
+ type: z.literal('text'),
381
423
  id: z.string(),
382
424
  status: z.enum(['cancelled', 'expired', 'failed', 'succeeded']),
383
425
  })
@@ -64,6 +64,7 @@ export type GatewayModelId =
64
64
  | 'deepseek/deepseek-v4-flash-vision-exp'
65
65
  | 'deepseek/deepseek-v4-pro'
66
66
  | 'deepseek/deepseek-v4-pro-0813'
67
+ | 'deepseek/deepseek-v4.1-flash-beta'
67
68
  | 'google/gemini-2.5-flash'
68
69
  | 'google/gemini-2.5-flash-image'
69
70
  | 'google/gemini-2.5-flash-lite'
@@ -88,6 +89,8 @@ export type GatewayModelId =
88
89
  | 'inclusionai/ling-3.0-flash'
89
90
  | 'inclusionai/ling-3.0-flash-fin'
90
91
  | 'inclusionai/ling-3.0-flash-fin-free'
92
+ | 'inclusionai/ling-3.0-flash-sante'
93
+ | 'inclusionai/ling-3.0-flash-sante-free'
91
94
  | 'interfaze/interfaze-beta'
92
95
  | 'kwaipilot/kat-coder-air-v2.5'
93
96
  | 'kwaipilot/kat-coder-pro-v1'
@@ -110,10 +113,8 @@ export type GatewayModelId =
110
113
  | 'minimax/minimax-m2.5'
111
114
  | 'minimax/minimax-m2.5-highspeed'
112
115
  | 'minimax/minimax-m2.7'
113
- | 'minimax/minimax-m2.7-free'
114
116
  | 'minimax/minimax-m2.7-highspeed'
115
117
  | 'minimax/minimax-m3'
116
- | 'minimax/minimax-m3-free'
117
118
  | 'mistral/codestral'
118
119
  | 'mistral/devstral-2'
119
120
  | 'mistral/devstral-small-2'
@@ -187,6 +188,8 @@ export type GatewayModelId =
187
188
  | 'openai/gpt-5.6-sol-fast'
188
189
  | 'openai/gpt-5.6-terra'
189
190
  | 'openai/gpt-5.6-terra-fast'
191
+ | 'openai/gpt-6-astra'
192
+ | 'openai/gpt-6-astra-fast'
190
193
  | 'openai/gpt-oss-120b'
191
194
  | 'openai/gpt-oss-20b'
192
195
  | 'openai/gpt-oss-safeguard-120b'
@@ -22,7 +22,7 @@ export type GatewayProviderOptions = {
22
22
  has?: Array<'implicit-caching' | 'vision'>;
23
23
 
24
24
  /**
25
- * Idempotency key for `experimental_startTextBatch`: retries with the same
25
+ * Idempotency key for `experimental_startBatch`: retries with the same
26
26
  * key replay the original batch instead of creating a duplicate.
27
27
  */
28
28
  idempotencyKey?: string;
@@ -31,7 +31,8 @@ import {
31
31
  type GatewayGenerationInfoParams,
32
32
  type GatewayGenerationInfo,
33
33
  } from './gateway-generation-info';
34
- import { GatewayBatchLanguageModel } from './gateway-language-model-batch';
34
+ import { GatewayBatch } from './gateway-batch';
35
+ import { GatewayLanguageModel } from './gateway-language-model';
35
36
  import { GatewayEmbeddingModel } from './gateway-embedding-model';
36
37
  import { GatewayImageModel } from './gateway-image-model';
37
38
  import { GatewayVideoModel } from './gateway-video-model';
@@ -54,7 +55,7 @@ import { getVercelOidcToken, getVercelRequestId } from './vercel-environment';
54
55
  import type { GatewayModelId } from './gateway-language-model-settings';
55
56
  import type {
56
57
  EmbeddingModelV4,
57
- Experimental_BatchLanguageModelV4 as BatchLanguageModelV4,
58
+ Experimental_BatchV4 as BatchV4,
58
59
  ImageModelV4,
59
60
  RerankingModelV4,
60
61
  SpeechModelV4,
@@ -63,21 +64,25 @@ import type {
63
64
  Experimental_RealtimeFactoryV4 as RealtimeFactoryV4,
64
65
  Experimental_RealtimeFactoryV4GetTokenOptions as RealtimeFactoryV4GetTokenOptions,
65
66
  ProviderV4,
67
+ LanguageModelV4,
66
68
  } from '@ai-sdk/provider';
67
69
  import { VERSION } from './version';
68
70
 
69
71
  export interface GatewayProvider extends ProviderV4 {
70
- (modelId: GatewayModelId): BatchLanguageModelV4;
72
+ (modelId: GatewayModelId): LanguageModelV4;
71
73
 
72
74
  /**
73
75
  * Creates a model for text generation.
74
76
  */
75
- chat(modelId: GatewayModelId): BatchLanguageModelV4;
77
+ chat(modelId: GatewayModelId): LanguageModelV4;
76
78
 
77
79
  /**
78
80
  * Creates a model for text generation.
79
81
  */
80
- languageModel(modelId: GatewayModelId): BatchLanguageModelV4;
82
+ languageModel(modelId: GatewayModelId): LanguageModelV4;
83
+
84
+ /** Returns a BatchV4 interface for processing batches with AI Gateway. */
85
+ experimental_batch(): BatchV4<{ text: GatewayModelId }>;
81
86
 
82
87
  /**
83
88
  * Returns available providers and models for use with the remote provider.
@@ -417,7 +422,7 @@ export function createGateway(
417
422
  };
418
423
 
419
424
  const createLanguageModel = (modelId: GatewayModelId) => {
420
- return new GatewayBatchLanguageModel(modelId, {
425
+ return new GatewayLanguageModel(modelId, {
421
426
  provider: 'gateway',
422
427
  baseURL,
423
428
  headers: getHeaders,
@@ -426,6 +431,15 @@ export function createGateway(
426
431
  });
427
432
  };
428
433
 
434
+ const createBatch = () =>
435
+ new GatewayBatch({
436
+ provider: 'gateway',
437
+ baseURL,
438
+ headers: getHeaders,
439
+ fetch: options.fetch,
440
+ o11yHeaders: createO11yHeaders(),
441
+ });
442
+
429
443
  const getAvailableModels = async () => {
430
444
  const now = options._internal?.currentDate?.().getTime() ?? Date.now();
431
445
  if (!pendingMetadata || now - lastFetchTime > cacheRefreshMillis) {
@@ -522,6 +536,7 @@ export function createGateway(
522
536
  });
523
537
  };
524
538
  provider.languageModel = createLanguageModel;
539
+ provider.experimental_batch = createBatch;
525
540
  const createEmbeddingModel = (modelId: GatewayEmbeddingModelId) => {
526
541
  return new GatewayEmbeddingModel(modelId, {
527
542
  provider: 'gateway',