@ai-sdk/gateway 4.0.55 → 4.0.57
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/dist/index.d.ts +43 -19
- package/dist/index.js +614 -217
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/errors/create-gateway-error.ts +8 -0
- package/src/errors/gateway-not-found-error.ts +35 -0
- package/src/errors/index.ts +1 -0
- package/src/gateway-language-model-batch.ts +526 -0
- package/src/gateway-language-model.ts +5 -3
- package/src/gateway-provider-options.ts +6 -0
- package/src/gateway-provider.ts +6 -6
- package/src/index.ts +1 -0
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ai-sdk/gateway",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "4.0.
|
|
4
|
+
"version": "4.0.57",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "Apache-2.0",
|
|
7
7
|
"sideEffects": false,
|
|
@@ -32,7 +32,7 @@
|
|
|
32
32
|
"dependencies": {
|
|
33
33
|
"@vercel/oidc": "3.2.0",
|
|
34
34
|
"@ai-sdk/provider": "4.0.7",
|
|
35
|
-
"@ai-sdk/provider-utils": "5.0.
|
|
35
|
+
"@ai-sdk/provider-utils": "5.0.28"
|
|
36
36
|
},
|
|
37
37
|
"devDependencies": {
|
|
38
38
|
"@types/node": "22.19.19",
|
|
@@ -7,6 +7,7 @@ import {
|
|
|
7
7
|
GatewayModelNotFoundError,
|
|
8
8
|
modelNotFoundParamSchema,
|
|
9
9
|
} from './gateway-model-not-found-error';
|
|
10
|
+
import { GatewayNotFoundError } from './gateway-not-found-error';
|
|
10
11
|
import { GatewayInternalServerError } from './gateway-internal-server-error';
|
|
11
12
|
import { GatewayFailedDependencyError } from './gateway-failed-dependency-error';
|
|
12
13
|
import {
|
|
@@ -99,6 +100,13 @@ export async function createGatewayErrorFromResponse({
|
|
|
99
100
|
generationId,
|
|
100
101
|
});
|
|
101
102
|
}
|
|
103
|
+
case 'not_found':
|
|
104
|
+
return new GatewayNotFoundError({
|
|
105
|
+
message,
|
|
106
|
+
statusCode,
|
|
107
|
+
cause,
|
|
108
|
+
generationId,
|
|
109
|
+
});
|
|
102
110
|
case 'internal_server_error':
|
|
103
111
|
return new GatewayInternalServerError({
|
|
104
112
|
message,
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import { GatewayError } from './gateway-error';
|
|
2
|
+
|
|
3
|
+
const name = 'GatewayNotFoundError';
|
|
4
|
+
const marker = `vercel.ai.gateway.error.${name}`;
|
|
5
|
+
const symbol = Symbol.for(marker);
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Not found - the requested Gateway resource does not exist or is not
|
|
9
|
+
* visible to the caller (e.g. an unknown async batch/video job id).
|
|
10
|
+
* Distinct from `GatewayModelNotFoundError`, which is model-specific.
|
|
11
|
+
*/
|
|
12
|
+
export class GatewayNotFoundError extends GatewayError {
|
|
13
|
+
private readonly [symbol] = true; // used in isInstance
|
|
14
|
+
|
|
15
|
+
readonly name = name;
|
|
16
|
+
readonly type = 'not_found';
|
|
17
|
+
|
|
18
|
+
constructor({
|
|
19
|
+
message = 'Resource not found',
|
|
20
|
+
statusCode = 404,
|
|
21
|
+
cause,
|
|
22
|
+
generationId,
|
|
23
|
+
}: {
|
|
24
|
+
message?: string;
|
|
25
|
+
statusCode?: number;
|
|
26
|
+
cause?: unknown;
|
|
27
|
+
generationId?: string;
|
|
28
|
+
} = {}) {
|
|
29
|
+
super({ message, statusCode, cause, generationId });
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
static isInstance(error: unknown): error is GatewayNotFoundError {
|
|
33
|
+
return GatewayError.hasMarker(error) && symbol in error;
|
|
34
|
+
}
|
|
35
|
+
}
|
package/src/errors/index.ts
CHANGED
|
@@ -14,6 +14,7 @@ export {
|
|
|
14
14
|
GatewayModelNotFoundError,
|
|
15
15
|
modelNotFoundParamSchema,
|
|
16
16
|
} from './gateway-model-not-found-error';
|
|
17
|
+
export { GatewayNotFoundError } from './gateway-not-found-error';
|
|
17
18
|
export { GatewayRateLimitError } from './gateway-rate-limit-error';
|
|
18
19
|
export { GatewayResponseError } from './gateway-response-error';
|
|
19
20
|
export { GatewayTimeoutError } from './gateway-timeout-error';
|
|
@@ -0,0 +1,526 @@
|
|
|
1
|
+
import {
|
|
2
|
+
APICallError,
|
|
3
|
+
type Experimental_BatchLanguageModelV4 as BatchLanguageModelV4,
|
|
4
|
+
type Experimental_BatchV4ItemResult as BatchV4ItemResult,
|
|
5
|
+
type Experimental_BatchV4OperationOptions as BatchV4OperationOptions,
|
|
6
|
+
type Experimental_BatchV4StartOptions as BatchV4StartOptions,
|
|
7
|
+
type Experimental_BatchV4StartResult as BatchV4StartResult,
|
|
8
|
+
type Experimental_BatchV4Status as BatchV4Status,
|
|
9
|
+
type Experimental_LanguageModelV4BatchRequest as LanguageModelV4BatchRequest,
|
|
10
|
+
type LanguageModelV4GenerateResult,
|
|
11
|
+
type SharedV4ProviderMetadata,
|
|
12
|
+
type SharedV4ProviderOptions,
|
|
13
|
+
} from '@ai-sdk/provider';
|
|
14
|
+
import {
|
|
15
|
+
combineHeaders,
|
|
16
|
+
convertAsyncIteratorToReadableStream,
|
|
17
|
+
createJsonErrorResponseHandler,
|
|
18
|
+
createJsonResponseHandler,
|
|
19
|
+
getErrorMessage,
|
|
20
|
+
parseJSON,
|
|
21
|
+
postJsonToApi,
|
|
22
|
+
resolve,
|
|
23
|
+
WORKFLOW_SERIALIZE,
|
|
24
|
+
WORKFLOW_DESERIALIZE,
|
|
25
|
+
} from '@ai-sdk/provider-utils';
|
|
26
|
+
import { z } from './zod';
|
|
27
|
+
import {
|
|
28
|
+
GatewayLanguageModel,
|
|
29
|
+
type GatewayChatConfig,
|
|
30
|
+
} from './gateway-language-model';
|
|
31
|
+
import type { GatewayModelId } from './gateway-language-model-settings';
|
|
32
|
+
import { asGatewayError } from './errors';
|
|
33
|
+
import { parseAuthMethod } from './errors/parse-auth-method';
|
|
34
|
+
|
|
35
|
+
export class GatewayBatchLanguageModel
|
|
36
|
+
extends GatewayLanguageModel
|
|
37
|
+
implements BatchLanguageModelV4
|
|
38
|
+
{
|
|
39
|
+
static [WORKFLOW_SERIALIZE](model: GatewayBatchLanguageModel) {
|
|
40
|
+
return GatewayLanguageModel[WORKFLOW_SERIALIZE](model);
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
static [WORKFLOW_DESERIALIZE](options: {
|
|
44
|
+
modelId: GatewayModelId;
|
|
45
|
+
config: GatewayChatConfig;
|
|
46
|
+
}) {
|
|
47
|
+
return new GatewayBatchLanguageModel(options.modelId, options.config);
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
constructor(modelId: GatewayModelId, config: GatewayChatConfig) {
|
|
51
|
+
super(modelId, config);
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Starts a durable batch of text-generation requests through the Gateway's
|
|
56
|
+
* async batch surface (`POST {baseURL}/batch/start`). The returned
|
|
57
|
+
* `batchId` is the Gateway job id — provider-native batch ids stay
|
|
58
|
+
* server-side, so status and results always route back through the
|
|
59
|
+
* Gateway job.
|
|
60
|
+
*/
|
|
61
|
+
async experimental_doStartBatch({
|
|
62
|
+
requests,
|
|
63
|
+
providerOptions,
|
|
64
|
+
headers,
|
|
65
|
+
abortSignal,
|
|
66
|
+
}: BatchV4StartOptions<LanguageModelV4BatchRequest>): Promise<BatchV4StartResult> {
|
|
67
|
+
const resolvedHeaders = this.config.headers
|
|
68
|
+
? await resolve(this.config.headers)
|
|
69
|
+
: undefined;
|
|
70
|
+
|
|
71
|
+
const idempotencyKey = getGatewayBatchIdempotencyKey(providerOptions);
|
|
72
|
+
const forwardedProviderOptions = omitGatewayIdempotencyKey(providerOptions);
|
|
73
|
+
|
|
74
|
+
try {
|
|
75
|
+
const { value: responseBody } = await postJsonToApi({
|
|
76
|
+
url: this.getBatchUrl('start'),
|
|
77
|
+
headers: combineHeaders(
|
|
78
|
+
resolvedHeaders,
|
|
79
|
+
headers,
|
|
80
|
+
this.getBatchConfigHeaders(),
|
|
81
|
+
await resolve(this.config.o11yHeaders),
|
|
82
|
+
idempotencyKey != null
|
|
83
|
+
? { 'idempotency-key': idempotencyKey }
|
|
84
|
+
: undefined,
|
|
85
|
+
),
|
|
86
|
+
body: {
|
|
87
|
+
modelId: this.modelId,
|
|
88
|
+
requests: requests.map(request => ({
|
|
89
|
+
id: request.id,
|
|
90
|
+
options: this.maybeEncodeFileParts(request.options),
|
|
91
|
+
})),
|
|
92
|
+
...(forwardedProviderOptions != null && {
|
|
93
|
+
providerOptions: forwardedProviderOptions,
|
|
94
|
+
}),
|
|
95
|
+
},
|
|
96
|
+
successfulResponseHandler: createJsonResponseHandler(
|
|
97
|
+
gatewayBatchStartResponseSchema,
|
|
98
|
+
),
|
|
99
|
+
failedResponseHandler: createJsonErrorResponseHandler({
|
|
100
|
+
errorSchema: z.any(),
|
|
101
|
+
errorToMessage: data => getErrorMessage(data) ?? 'unknown error',
|
|
102
|
+
}),
|
|
103
|
+
...(abortSignal && { abortSignal }),
|
|
104
|
+
fetch: this.config.fetch,
|
|
105
|
+
});
|
|
106
|
+
|
|
107
|
+
return {
|
|
108
|
+
batchId: responseBody.batchId,
|
|
109
|
+
...convertGatewayBatchStatus(responseBody),
|
|
110
|
+
warnings: (responseBody.warnings ??
|
|
111
|
+
[]) as unknown as BatchV4StartResult['warnings'],
|
|
112
|
+
};
|
|
113
|
+
} catch (error) {
|
|
114
|
+
// Preserve cancellation: an aborted batch start may still have been
|
|
115
|
+
// accepted server-side, so it must not surface as a retryable 500.
|
|
116
|
+
if (isAbortOrTimeoutError(error)) {
|
|
117
|
+
throw error;
|
|
118
|
+
}
|
|
119
|
+
throw await asGatewayError(
|
|
120
|
+
error,
|
|
121
|
+
await parseAuthMethod(resolvedHeaders ?? {}),
|
|
122
|
+
);
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Retrieves the lifecycle status of a Gateway batch job
|
|
128
|
+
* (`POST {baseURL}/batch/status`).
|
|
129
|
+
*/
|
|
130
|
+
async experimental_doGetBatchStatus({
|
|
131
|
+
batchId,
|
|
132
|
+
headers,
|
|
133
|
+
abortSignal,
|
|
134
|
+
}: BatchV4OperationOptions): Promise<BatchV4Status> {
|
|
135
|
+
const resolvedHeaders = this.config.headers
|
|
136
|
+
? await resolve(this.config.headers)
|
|
137
|
+
: undefined;
|
|
138
|
+
|
|
139
|
+
try {
|
|
140
|
+
const { value: responseBody } = await postJsonToApi({
|
|
141
|
+
url: this.getBatchUrl('status'),
|
|
142
|
+
headers: combineHeaders(
|
|
143
|
+
resolvedHeaders,
|
|
144
|
+
headers,
|
|
145
|
+
this.getBatchConfigHeaders(),
|
|
146
|
+
await resolve(this.config.o11yHeaders),
|
|
147
|
+
),
|
|
148
|
+
body: { batchId },
|
|
149
|
+
successfulResponseHandler: createJsonResponseHandler(
|
|
150
|
+
gatewayBatchStatusResponseSchema,
|
|
151
|
+
),
|
|
152
|
+
failedResponseHandler: createJsonErrorResponseHandler({
|
|
153
|
+
errorSchema: z.any(),
|
|
154
|
+
errorToMessage: data => getErrorMessage(data) ?? 'unknown error',
|
|
155
|
+
}),
|
|
156
|
+
...(abortSignal && { abortSignal }),
|
|
157
|
+
fetch: this.config.fetch,
|
|
158
|
+
});
|
|
159
|
+
|
|
160
|
+
return convertGatewayBatchStatus(responseBody);
|
|
161
|
+
} catch (error) {
|
|
162
|
+
if (isAbortOrTimeoutError(error)) {
|
|
163
|
+
throw error;
|
|
164
|
+
}
|
|
165
|
+
throw await asGatewayError(
|
|
166
|
+
error,
|
|
167
|
+
await parseAuthMethod(resolvedHeaders ?? {}),
|
|
168
|
+
);
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Streams the per-request results of a terminal Gateway batch job
|
|
174
|
+
* (`POST {baseURL}/batch/results`, `application/x-ndjson`: one
|
|
175
|
+
* `BatchV4ItemResult` JSON object per line). Items are validated minimally
|
|
176
|
+
* (id + status) and passed through — the Gateway sanitizes them
|
|
177
|
+
* server-side. The route responds 400 while the batch is non-terminal.
|
|
178
|
+
*/
|
|
179
|
+
async experimental_doGetBatchResults({
|
|
180
|
+
batchId,
|
|
181
|
+
headers,
|
|
182
|
+
abortSignal,
|
|
183
|
+
}: BatchV4OperationOptions): Promise<
|
|
184
|
+
ReadableStream<BatchV4ItemResult<LanguageModelV4GenerateResult>>
|
|
185
|
+
> {
|
|
186
|
+
const resolvedHeaders = this.config.headers
|
|
187
|
+
? await resolve(this.config.headers)
|
|
188
|
+
: undefined;
|
|
189
|
+
|
|
190
|
+
try {
|
|
191
|
+
const { value: stream } = await postJsonToApi({
|
|
192
|
+
url: this.getBatchUrl('results'),
|
|
193
|
+
headers: combineHeaders(
|
|
194
|
+
resolvedHeaders,
|
|
195
|
+
headers,
|
|
196
|
+
this.getBatchConfigHeaders(),
|
|
197
|
+
await resolve(this.config.o11yHeaders),
|
|
198
|
+
),
|
|
199
|
+
body: { batchId },
|
|
200
|
+
successfulResponseHandler: async ({
|
|
201
|
+
response,
|
|
202
|
+
url,
|
|
203
|
+
requestBodyValues,
|
|
204
|
+
}: {
|
|
205
|
+
url: string;
|
|
206
|
+
requestBodyValues: unknown;
|
|
207
|
+
response: Response;
|
|
208
|
+
}) => {
|
|
209
|
+
if (response.body == null) {
|
|
210
|
+
throw new APICallError({
|
|
211
|
+
message: 'Batch results response body is empty',
|
|
212
|
+
url,
|
|
213
|
+
requestBodyValues,
|
|
214
|
+
statusCode: response.status,
|
|
215
|
+
});
|
|
216
|
+
}
|
|
217
|
+
return {
|
|
218
|
+
value: response.body,
|
|
219
|
+
responseHeaders: Object.fromEntries([...response.headers]),
|
|
220
|
+
};
|
|
221
|
+
},
|
|
222
|
+
failedResponseHandler: createJsonErrorResponseHandler({
|
|
223
|
+
errorSchema: z.any(),
|
|
224
|
+
errorToMessage: data => getErrorMessage(data) ?? 'unknown error',
|
|
225
|
+
}),
|
|
226
|
+
...(abortSignal && { abortSignal }),
|
|
227
|
+
fetch: this.config.fetch,
|
|
228
|
+
});
|
|
229
|
+
|
|
230
|
+
return convertAsyncIteratorToReadableStream(
|
|
231
|
+
parseGatewayBatchResultLines(stream),
|
|
232
|
+
);
|
|
233
|
+
} catch (error) {
|
|
234
|
+
if (isAbortOrTimeoutError(error)) {
|
|
235
|
+
throw error;
|
|
236
|
+
}
|
|
237
|
+
throw await asGatewayError(
|
|
238
|
+
error,
|
|
239
|
+
await parseAuthMethod(resolvedHeaders ?? {}),
|
|
240
|
+
);
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
private getBatchUrl(path: 'results' | 'start' | 'status') {
|
|
245
|
+
return `${this.config.baseURL}/batch/${path}`;
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
private getBatchConfigHeaders() {
|
|
249
|
+
return {
|
|
250
|
+
'ai-model-id': this.modelId,
|
|
251
|
+
};
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
/**
|
|
256
|
+
* Extracts the optional Gateway idempotency key from
|
|
257
|
+
* `providerOptions.gateway.idempotencyKey`. It is sent as the
|
|
258
|
+
* `idempotency-key` request header — the Gateway's replay contract for batch
|
|
259
|
+
* starts — and stripped from the forwarded body by
|
|
260
|
+
* `omitGatewayIdempotencyKey`.
|
|
261
|
+
*/
|
|
262
|
+
function getGatewayBatchIdempotencyKey(
|
|
263
|
+
providerOptions: SharedV4ProviderOptions | undefined,
|
|
264
|
+
): string | undefined {
|
|
265
|
+
const gatewayOptions = providerOptions?.gateway;
|
|
266
|
+
if (
|
|
267
|
+
gatewayOptions == null ||
|
|
268
|
+
typeof gatewayOptions !== 'object' ||
|
|
269
|
+
Array.isArray(gatewayOptions)
|
|
270
|
+
) {
|
|
271
|
+
return undefined;
|
|
272
|
+
}
|
|
273
|
+
const key = (gatewayOptions as { idempotencyKey?: unknown }).idempotencyKey;
|
|
274
|
+
return typeof key === 'string' && key.length > 0 ? key : undefined;
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
/**
|
|
278
|
+
* Removes `gateway.idempotencyKey` from the providerOptions forwarded in the
|
|
279
|
+
* request body. The key is transport metadata (it rides the `idempotency-key`
|
|
280
|
+
* header); the Gateway hashes the raw body for replay payload identity but
|
|
281
|
+
* normalizes the header separately, so keeping it out of the payload prevents
|
|
282
|
+
* equivalent retries from producing different digests (a false 422).
|
|
283
|
+
*/
|
|
284
|
+
function omitGatewayIdempotencyKey(
|
|
285
|
+
providerOptions: SharedV4ProviderOptions | undefined,
|
|
286
|
+
): SharedV4ProviderOptions | undefined {
|
|
287
|
+
const gatewayOptions = providerOptions?.gateway;
|
|
288
|
+
if (
|
|
289
|
+
gatewayOptions == null ||
|
|
290
|
+
typeof gatewayOptions !== 'object' ||
|
|
291
|
+
Array.isArray(gatewayOptions) ||
|
|
292
|
+
!('idempotencyKey' in gatewayOptions)
|
|
293
|
+
) {
|
|
294
|
+
return providerOptions;
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
const { idempotencyKey: _idempotencyKey, ...restGatewayOptions } =
|
|
298
|
+
gatewayOptions as Record<string, unknown>;
|
|
299
|
+
const restProviderOptions: Record<string, unknown> = { ...providerOptions };
|
|
300
|
+
if (Object.keys(restGatewayOptions).length === 0) {
|
|
301
|
+
delete restProviderOptions.gateway;
|
|
302
|
+
} else {
|
|
303
|
+
restProviderOptions.gateway = restGatewayOptions;
|
|
304
|
+
}
|
|
305
|
+
if (Object.keys(restProviderOptions).length === 0) {
|
|
306
|
+
return undefined;
|
|
307
|
+
}
|
|
308
|
+
return restProviderOptions as SharedV4ProviderOptions;
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/**
|
|
312
|
+
* Matches cancellation errors (`AbortError`/`TimeoutError`), which
|
|
313
|
+
* `asGatewayError` would otherwise wrap into a retryable Gateway 500. Kept
|
|
314
|
+
* local because `isAbortError` is not exported from `@ai-sdk/provider-utils`;
|
|
315
|
+
* `DOMException` does not extend `Error`, so both must be checked.
|
|
316
|
+
*/
|
|
317
|
+
function isAbortOrTimeoutError(error: unknown): boolean {
|
|
318
|
+
if (!(error instanceof Error || error instanceof DOMException)) {
|
|
319
|
+
return false;
|
|
320
|
+
}
|
|
321
|
+
return error.name === 'AbortError' || error.name === 'TimeoutError';
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
function convertGatewayBatchStatus(body: {
|
|
325
|
+
status: 'completed' | 'failed' | 'pending';
|
|
326
|
+
rawStatus?: string | null;
|
|
327
|
+
requestCounts?: {
|
|
328
|
+
total?: number | null;
|
|
329
|
+
pending?: number | null;
|
|
330
|
+
completed?: number | null;
|
|
331
|
+
failed?: number | null;
|
|
332
|
+
} | null;
|
|
333
|
+
error?: {
|
|
334
|
+
message: string;
|
|
335
|
+
type?: string | null;
|
|
336
|
+
code?: string | null;
|
|
337
|
+
statusCode?: number | null;
|
|
338
|
+
} | null;
|
|
339
|
+
createdAt?: string | null;
|
|
340
|
+
expiresAt?: string | null;
|
|
341
|
+
providerMetadata?: Record<string, Record<string, unknown>> | null;
|
|
342
|
+
}): BatchV4Status {
|
|
343
|
+
const requestCounts = convertGatewayBatchRequestCounts(body.requestCounts);
|
|
344
|
+
|
|
345
|
+
return {
|
|
346
|
+
status: body.status,
|
|
347
|
+
...(body.rawStatus != null && { rawStatus: body.rawStatus }),
|
|
348
|
+
...(requestCounts != null && { requestCounts }),
|
|
349
|
+
...(body.error != null && {
|
|
350
|
+
error: {
|
|
351
|
+
message: body.error.message,
|
|
352
|
+
...(body.error.type != null && { type: body.error.type }),
|
|
353
|
+
...(body.error.code != null && { code: body.error.code }),
|
|
354
|
+
...(body.error.statusCode != null && {
|
|
355
|
+
statusCode: body.error.statusCode,
|
|
356
|
+
}),
|
|
357
|
+
},
|
|
358
|
+
}),
|
|
359
|
+
...(body.createdAt != null && { createdAt: body.createdAt }),
|
|
360
|
+
...(body.expiresAt != null && { expiresAt: body.expiresAt }),
|
|
361
|
+
...(body.providerMetadata != null && {
|
|
362
|
+
providerMetadata: body.providerMetadata as SharedV4ProviderMetadata,
|
|
363
|
+
}),
|
|
364
|
+
};
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
/**
|
|
368
|
+
* The spec's `requestCounts` requires all four counters; the Gateway's
|
|
369
|
+
* persisted descriptor allows partial counts. Only forward counts when the
|
|
370
|
+
* full set is present rather than fabricating zeros.
|
|
371
|
+
*/
|
|
372
|
+
function convertGatewayBatchRequestCounts(
|
|
373
|
+
counts:
|
|
374
|
+
| {
|
|
375
|
+
total?: number | null;
|
|
376
|
+
pending?: number | null;
|
|
377
|
+
completed?: number | null;
|
|
378
|
+
failed?: number | null;
|
|
379
|
+
}
|
|
380
|
+
| null
|
|
381
|
+
| undefined,
|
|
382
|
+
): BatchV4Status['requestCounts'] | undefined {
|
|
383
|
+
if (
|
|
384
|
+
counts == null ||
|
|
385
|
+
typeof counts.total !== 'number' ||
|
|
386
|
+
typeof counts.pending !== 'number' ||
|
|
387
|
+
typeof counts.completed !== 'number' ||
|
|
388
|
+
typeof counts.failed !== 'number'
|
|
389
|
+
) {
|
|
390
|
+
return undefined;
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
return {
|
|
394
|
+
total: counts.total,
|
|
395
|
+
pending: counts.pending,
|
|
396
|
+
completed: counts.completed,
|
|
397
|
+
failed: counts.failed,
|
|
398
|
+
};
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
/**
|
|
402
|
+
* Incremental NDJSON line splitter for the batch results stream: buffers
|
|
403
|
+
* partial lines across chunks and flushes a trailing line without a final
|
|
404
|
+
* newline. Each non-empty line is one `BatchV4ItemResult` JSON object.
|
|
405
|
+
*
|
|
406
|
+
* @param stream - The raw NDJSON byte stream from the batch results route.
|
|
407
|
+
* @yields One minimally-validated `BatchV4ItemResult` per non-empty line.
|
|
408
|
+
*/
|
|
409
|
+
async function* parseGatewayBatchResultLines(
|
|
410
|
+
stream: ReadableStream<Uint8Array>,
|
|
411
|
+
): AsyncGenerator<BatchV4ItemResult<LanguageModelV4GenerateResult>> {
|
|
412
|
+
const reader = stream.getReader();
|
|
413
|
+
const decoder = new TextDecoder();
|
|
414
|
+
let buffer = '';
|
|
415
|
+
let finished = false;
|
|
416
|
+
|
|
417
|
+
try {
|
|
418
|
+
while (true) {
|
|
419
|
+
const { done, value } = await reader.read();
|
|
420
|
+
|
|
421
|
+
if (done) {
|
|
422
|
+
finished = true;
|
|
423
|
+
buffer += decoder.decode();
|
|
424
|
+
break;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
buffer += decoder.decode(value, { stream: true });
|
|
428
|
+
|
|
429
|
+
let lineEnd = buffer.indexOf('\n');
|
|
430
|
+
while (lineEnd !== -1) {
|
|
431
|
+
const line = buffer.slice(0, lineEnd).replace(/\r$/, '');
|
|
432
|
+
buffer = buffer.slice(lineEnd + 1);
|
|
433
|
+
|
|
434
|
+
if (line.trim().length > 0) {
|
|
435
|
+
yield await parseGatewayBatchResultLine(line);
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
lineEnd = buffer.indexOf('\n');
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
const finalLine = buffer.replace(/\r$/, '');
|
|
443
|
+
if (finalLine.trim().length > 0) {
|
|
444
|
+
yield await parseGatewayBatchResultLine(finalLine);
|
|
445
|
+
}
|
|
446
|
+
} finally {
|
|
447
|
+
if (!finished) {
|
|
448
|
+
await reader.cancel().catch(() => {});
|
|
449
|
+
}
|
|
450
|
+
reader.releaseLock();
|
|
451
|
+
}
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
async function parseGatewayBatchResultLine(
|
|
455
|
+
line: string,
|
|
456
|
+
): Promise<BatchV4ItemResult<LanguageModelV4GenerateResult>> {
|
|
457
|
+
// Minimal validation (id + status); items pass through otherwise — the
|
|
458
|
+
// Gateway already sanitizes them server-side.
|
|
459
|
+
const parsed = await parseJSON({
|
|
460
|
+
text: line,
|
|
461
|
+
schema: gatewayBatchItemResultLineSchema,
|
|
462
|
+
});
|
|
463
|
+
const item =
|
|
464
|
+
parsed as unknown as BatchV4ItemResult<LanguageModelV4GenerateResult>;
|
|
465
|
+
// JSON carries `response.timestamp` as an ISO string; core expects a Date
|
|
466
|
+
// (`GeneratedFile`-style consumers call `.toISOString()`).
|
|
467
|
+
if (item.status === 'succeeded') {
|
|
468
|
+
const response = item.result?.response;
|
|
469
|
+
if (response !== undefined && typeof response.timestamp === 'string') {
|
|
470
|
+
response.timestamp = new Date(response.timestamp);
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
return item;
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
const gatewayBatchItemResultLineSchema = z
|
|
477
|
+
.object({
|
|
478
|
+
id: z.string(),
|
|
479
|
+
status: z.enum(['cancelled', 'expired', 'failed', 'succeeded']),
|
|
480
|
+
})
|
|
481
|
+
.catchall(z.unknown());
|
|
482
|
+
|
|
483
|
+
const gatewayBatchErrorSchema = z.object({
|
|
484
|
+
message: z.string(),
|
|
485
|
+
type: z.string().nullish(),
|
|
486
|
+
code: z.string().nullish(),
|
|
487
|
+
statusCode: z.number().nullish(),
|
|
488
|
+
});
|
|
489
|
+
|
|
490
|
+
const gatewayBatchRequestCountsSchema = z.object({
|
|
491
|
+
total: z.number().nullish(),
|
|
492
|
+
pending: z.number().nullish(),
|
|
493
|
+
completed: z.number().nullish(),
|
|
494
|
+
failed: z.number().nullish(),
|
|
495
|
+
});
|
|
496
|
+
|
|
497
|
+
const gatewayBatchProviderMetadataSchema = z.record(
|
|
498
|
+
z.string(),
|
|
499
|
+
z.record(z.string(), z.unknown()),
|
|
500
|
+
);
|
|
501
|
+
|
|
502
|
+
const gatewayBatchStatusFieldsSchema = z.object({
|
|
503
|
+
status: z.enum(['completed', 'failed', 'pending']),
|
|
504
|
+
rawStatus: z.string().nullish(),
|
|
505
|
+
requestCounts: gatewayBatchRequestCountsSchema.nullish(),
|
|
506
|
+
error: gatewayBatchErrorSchema.nullish(),
|
|
507
|
+
createdAt: z.string().nullish(),
|
|
508
|
+
expiresAt: z.string().nullish(),
|
|
509
|
+
providerMetadata: gatewayBatchProviderMetadataSchema.nullish(),
|
|
510
|
+
});
|
|
511
|
+
|
|
512
|
+
const gatewayBatchStartResponseSchema = gatewayBatchStatusFieldsSchema.extend({
|
|
513
|
+
batchId: z.string(),
|
|
514
|
+
warnings: z
|
|
515
|
+
.array(
|
|
516
|
+
z
|
|
517
|
+
.object({
|
|
518
|
+
requestId: z.string().nullish(),
|
|
519
|
+
warning: z.unknown(),
|
|
520
|
+
})
|
|
521
|
+
.catchall(z.unknown()),
|
|
522
|
+
)
|
|
523
|
+
.nullish(),
|
|
524
|
+
});
|
|
525
|
+
|
|
526
|
+
const gatewayBatchStatusResponseSchema = gatewayBatchStatusFieldsSchema;
|
|
@@ -25,7 +25,7 @@ import type { GatewayModelId } from './gateway-language-model-settings';
|
|
|
25
25
|
import { asGatewayError } from './errors';
|
|
26
26
|
import { parseAuthMethod } from './errors/parse-auth-method';
|
|
27
27
|
|
|
28
|
-
type GatewayChatConfig = GatewayConfig & {
|
|
28
|
+
export type GatewayChatConfig = GatewayConfig & {
|
|
29
29
|
provider: string;
|
|
30
30
|
o11yHeaders: Resolvable<Record<string, string>>;
|
|
31
31
|
};
|
|
@@ -50,7 +50,7 @@ export class GatewayLanguageModel implements LanguageModelV4 {
|
|
|
50
50
|
|
|
51
51
|
constructor(
|
|
52
52
|
readonly modelId: GatewayModelId,
|
|
53
|
-
|
|
53
|
+
protected readonly config: GatewayChatConfig,
|
|
54
54
|
) {}
|
|
55
55
|
|
|
56
56
|
get provider(): string {
|
|
@@ -196,7 +196,9 @@ export class GatewayLanguageModel implements LanguageModelV4 {
|
|
|
196
196
|
* @param options - The options to encode.
|
|
197
197
|
* @returns The options with the file data encoded.
|
|
198
198
|
*/
|
|
199
|
-
|
|
199
|
+
protected maybeEncodeFileParts<
|
|
200
|
+
T extends Pick<LanguageModelV4CallOptions, 'prompt'>,
|
|
201
|
+
>(options: T): T {
|
|
200
202
|
for (const message of options.prompt) {
|
|
201
203
|
if (!Array.isArray(message.content)) {
|
|
202
204
|
continue;
|
|
@@ -21,6 +21,12 @@ export type GatewayProviderOptions = {
|
|
|
21
21
|
*/
|
|
22
22
|
has?: Array<'implicit-caching' | 'vision'>;
|
|
23
23
|
|
|
24
|
+
/**
|
|
25
|
+
* Idempotency key for `experimental_startTextBatch`: retries with the same
|
|
26
|
+
* key replay the original batch instead of creating a duplicate.
|
|
27
|
+
*/
|
|
28
|
+
idempotencyKey?: string;
|
|
29
|
+
|
|
24
30
|
/** Array of model slugs specifying fallback models to use in order. */
|
|
25
31
|
models?: string[];
|
|
26
32
|
|
package/src/gateway-provider.ts
CHANGED
|
@@ -31,7 +31,7 @@ import {
|
|
|
31
31
|
type GatewayGenerationInfoParams,
|
|
32
32
|
type GatewayGenerationInfo,
|
|
33
33
|
} from './gateway-generation-info';
|
|
34
|
-
import {
|
|
34
|
+
import { GatewayBatchLanguageModel } from './gateway-language-model-batch';
|
|
35
35
|
import { GatewayEmbeddingModel } from './gateway-embedding-model';
|
|
36
36
|
import { GatewayImageModel } from './gateway-image-model';
|
|
37
37
|
import { GatewayVideoModel } from './gateway-video-model';
|
|
@@ -53,8 +53,8 @@ import { gatewayTools } from './gateway-tools';
|
|
|
53
53
|
import { getVercelOidcToken, getVercelRequestId } from './vercel-environment';
|
|
54
54
|
import type { GatewayModelId } from './gateway-language-model-settings';
|
|
55
55
|
import type {
|
|
56
|
-
LanguageModelV4,
|
|
57
56
|
EmbeddingModelV4,
|
|
57
|
+
Experimental_BatchLanguageModelV4 as BatchLanguageModelV4,
|
|
58
58
|
ImageModelV4,
|
|
59
59
|
RerankingModelV4,
|
|
60
60
|
SpeechModelV4,
|
|
@@ -67,17 +67,17 @@ import type {
|
|
|
67
67
|
import { VERSION } from './version';
|
|
68
68
|
|
|
69
69
|
export interface GatewayProvider extends ProviderV4 {
|
|
70
|
-
(modelId: GatewayModelId):
|
|
70
|
+
(modelId: GatewayModelId): BatchLanguageModelV4;
|
|
71
71
|
|
|
72
72
|
/**
|
|
73
73
|
* Creates a model for text generation.
|
|
74
74
|
*/
|
|
75
|
-
chat(modelId: GatewayModelId):
|
|
75
|
+
chat(modelId: GatewayModelId): BatchLanguageModelV4;
|
|
76
76
|
|
|
77
77
|
/**
|
|
78
78
|
* Creates a model for text generation.
|
|
79
79
|
*/
|
|
80
|
-
languageModel(modelId: GatewayModelId):
|
|
80
|
+
languageModel(modelId: GatewayModelId): BatchLanguageModelV4;
|
|
81
81
|
|
|
82
82
|
/**
|
|
83
83
|
* Returns available providers and models for use with the remote provider.
|
|
@@ -417,7 +417,7 @@ export function createGateway(
|
|
|
417
417
|
};
|
|
418
418
|
|
|
419
419
|
const createLanguageModel = (modelId: GatewayModelId) => {
|
|
420
|
-
return new
|
|
420
|
+
return new GatewayBatchLanguageModel(modelId, {
|
|
421
421
|
provider: 'gateway',
|
|
422
422
|
baseURL,
|
|
423
423
|
headers: getHeaders,
|