@alvin0/ai-agent-sdk-provider-http 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +348 -45
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +902 -402
- package/dist/index.js.map +1 -1
- package/package.json +3 -3
package/dist/index.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
import { CONTEXT_WINDOW_EXCEEDED_CODE, MODEL_ERROR_CODES, ModelAdapter, ModelError, ProviderRequestId, QUOTA_EXCEEDED_CODE, assertUsableApiKey, attributionHeaders, contentHasImage, createOperationId, detachedFrozen, isContextWindowExceededError, isQuotaExceededError, isSpanId, isTraceId, resolveRetryPolicy, safeErrorRecord, validateUsageCounters, waitForSettlement } from "@alvin0/ai-agent-sdk-core";
|
|
1
|
+
import { CONTEXT_WINDOW_EXCEEDED_CODE, MODEL_ERROR_CODES, ModelAdapter, ModelError, ProviderRequestId, QUOTA_EXCEEDED_CODE, assertUsableApiKey, attributionHeaders, contentHasDocument, contentHasImage, createOperationId, detachedFrozen, isContextWindowExceededError, isQuotaExceededError, isSpanId, isTraceId, resolveRetryPolicy, safeErrorRecord, validateUsageCounters, waitForSettlement } from "@alvin0/ai-agent-sdk-core";
|
|
2
2
|
import { createParser } from "eventsource-parser";
|
|
3
3
|
import { AgentSdkError, CREDENTIAL_CAPABILITY_API_VERSION } from "@alvin0/ai-agent-sdk-core/provider";
|
|
4
|
+
import { unknownEmbeddingModel } from "@alvin0/ai-agent-sdk-core/embedding";
|
|
4
5
|
|
|
5
6
|
//#region src/common/config.ts
|
|
6
7
|
/** Runtime wire-protocol contract version supported by this package. */
|
|
@@ -14,6 +15,7 @@ const HTTP_PROVIDER_ERROR_CODES = Object.freeze({
|
|
|
14
15
|
WIRE_BODY_INVALID: "HTTP_WIRE_BODY_INVALID",
|
|
15
16
|
WIRE_BODY_TOO_LARGE: "HTTP_WIRE_BODY_TOO_LARGE",
|
|
16
17
|
STREAM_MEDIA_TYPE_INVALID: "HTTP_STREAM_MEDIA_TYPE_INVALID",
|
|
18
|
+
JSON_MEDIA_TYPE_INVALID: "HTTP_JSON_MEDIA_TYPE_INVALID",
|
|
17
19
|
SSE_LIMIT_EXCEEDED: "HTTP_SSE_LIMIT_EXCEEDED",
|
|
18
20
|
REDIRECT_REJECTED: "HTTP_REDIRECT_REJECTED"
|
|
19
21
|
});
|
|
@@ -216,83 +218,6 @@ async function* requireTerminalFinish(source, displayName) {
|
|
|
216
218
|
yield finish;
|
|
217
219
|
}
|
|
218
220
|
|
|
219
|
-
//#endregion
|
|
220
|
-
//#region src/common/failure.ts
|
|
221
|
-
const ENCODER$1 = new TextEncoder();
|
|
222
|
-
const INVALID_FIELD = Symbol("invalid failure field");
|
|
223
|
-
/** Read an own data property without invoking getters or inherited state. */
|
|
224
|
-
function ownDataProbe(source, key) {
|
|
225
|
-
try {
|
|
226
|
-
const descriptor = Object.getOwnPropertyDescriptor(source, key);
|
|
227
|
-
if (descriptor === void 0) return {
|
|
228
|
-
present: false,
|
|
229
|
-
data: true
|
|
230
|
-
};
|
|
231
|
-
if (!("value" in descriptor)) return {
|
|
232
|
-
present: true,
|
|
233
|
-
data: false
|
|
234
|
-
};
|
|
235
|
-
return {
|
|
236
|
-
present: true,
|
|
237
|
-
data: true,
|
|
238
|
-
value: descriptor.value
|
|
239
|
-
};
|
|
240
|
-
} catch {
|
|
241
|
-
return {
|
|
242
|
-
present: true,
|
|
243
|
-
data: false
|
|
244
|
-
};
|
|
245
|
-
}
|
|
246
|
-
}
|
|
247
|
-
function boundedString(value, maxBytes) {
|
|
248
|
-
return typeof value === "string" && value.length > 0 && ENCODER$1.encode(value).byteLength <= maxBytes;
|
|
249
|
-
}
|
|
250
|
-
function optionalFailureField(source, key) {
|
|
251
|
-
const field = ownDataProbe(source, key);
|
|
252
|
-
return field.data ? field.value : INVALID_FIELD;
|
|
253
|
-
}
|
|
254
|
-
/**
|
|
255
|
-
* Validate the data twin carried by a ModelError from another core copy/realm.
|
|
256
|
-
* A lone outer code is deliberately insufficient: retry policy may trust a code
|
|
257
|
-
* only when the bounded inner envelope exists and agrees with it.
|
|
258
|
-
*/
|
|
259
|
-
function probeFailureEnvelope(value) {
|
|
260
|
-
if (typeof value !== "object" && typeof value !== "function" || value === null) return { kind: "absent" };
|
|
261
|
-
const outerCode = ownDataProbe(value, "code");
|
|
262
|
-
const carried = ownDataProbe(value, "failure");
|
|
263
|
-
if (!outerCode.present && !carried.present) return { kind: "absent" };
|
|
264
|
-
if (!outerCode.data || !carried.data || !boundedString(outerCode.value, HTTP_FOREIGN_FAILURE_LIMITS.codeBytes) || typeof carried.value !== "object" || carried.value === null || Array.isArray(carried.value)) return { kind: "invalid" };
|
|
265
|
-
const message = optionalFailureField(carried.value, "message");
|
|
266
|
-
const code = optionalFailureField(carried.value, "code");
|
|
267
|
-
const status = optionalFailureField(carried.value, "status");
|
|
268
|
-
const providerRetryAfterMs = optionalFailureField(carried.value, "providerRetryAfterMs");
|
|
269
|
-
const requestId = optionalFailureField(carried.value, "requestId");
|
|
270
|
-
if (message === INVALID_FIELD || code === INVALID_FIELD || status === INVALID_FIELD || providerRetryAfterMs === INVALID_FIELD || requestId === INVALID_FIELD || !boundedString(message, HTTP_FOREIGN_FAILURE_LIMITS.messageBytes) || !boundedString(code, HTTP_FOREIGN_FAILURE_LIMITS.codeBytes) || code !== outerCode.value || status !== void 0 && (!Number.isSafeInteger(status) || status < 100 || status > 599) || providerRetryAfterMs !== void 0 && (!Number.isFinite(providerRetryAfterMs) || providerRetryAfterMs <= 0) || requestId !== void 0 && !boundedString(requestId, HTTP_FOREIGN_FAILURE_LIMITS.requestIdBytes)) return { kind: "invalid" };
|
|
271
|
-
return {
|
|
272
|
-
kind: "valid",
|
|
273
|
-
failure: Object.freeze({
|
|
274
|
-
message,
|
|
275
|
-
code,
|
|
276
|
-
...status === void 0 ? {} : { status },
|
|
277
|
-
...providerRetryAfterMs === void 0 ? {} : { providerRetryAfterMs },
|
|
278
|
-
...requestId === void 0 ? {} : { requestId }
|
|
279
|
-
})
|
|
280
|
-
};
|
|
281
|
-
}
|
|
282
|
-
/** Normalize a foreign provider failure without relying on package class identity. */
|
|
283
|
-
function normalizeHttpBoundaryError(value, fallbackMessage) {
|
|
284
|
-
const envelope = probeFailureEnvelope(value);
|
|
285
|
-
if (envelope.kind === "absent") return new ModelError(fallbackMessage, MODEL_ERROR_CODES.TRANSPORT, { cause: value });
|
|
286
|
-
if (envelope.kind === "invalid") return new ModelError("provider supplied an invalid failure envelope", MODEL_ERROR_CODES.UNKNOWN, { cause: value });
|
|
287
|
-
const failure = envelope.failure;
|
|
288
|
-
return new ModelError(failure.message, failure.code, {
|
|
289
|
-
cause: value,
|
|
290
|
-
...failure.status === void 0 ? {} : { status: failure.status },
|
|
291
|
-
...failure.providerRetryAfterMs === void 0 ? {} : { providerRetryAfterMs: failure.providerRetryAfterMs },
|
|
292
|
-
...failure.requestId === void 0 ? {} : { requestId: failure.requestId }
|
|
293
|
-
});
|
|
294
|
-
}
|
|
295
|
-
|
|
296
221
|
//#endregion
|
|
297
222
|
//#region src/common/header-layers.ts
|
|
298
223
|
const DEFAULT_TRANSPORT_HEADERS = Object.freeze({
|
|
@@ -379,122 +304,122 @@ function headerError(message, key) {
|
|
|
379
304
|
}
|
|
380
305
|
|
|
381
306
|
//#endregion
|
|
382
|
-
//#region src/
|
|
307
|
+
//#region src/transport/connection.ts
|
|
383
308
|
/**
|
|
384
|
-
* The
|
|
309
|
+
* The connection snapshot every HTTP pipeline in this package shares.
|
|
385
310
|
*
|
|
386
|
-
*
|
|
387
|
-
*
|
|
388
|
-
*
|
|
389
|
-
*
|
|
390
|
-
* in the status code and both change what the caller should do, so getting them
|
|
391
|
-
* right once is worth more than getting them right three times.
|
|
311
|
+
* The snapshot exists to close a specific gap: if the endpoint and the credential
|
|
312
|
+
* were read separately, a configuration change between the two reads would send
|
|
313
|
+
* one generation's secret to another generation's URL. Reading them together, once
|
|
314
|
+
* per operation, makes that impossible.
|
|
392
315
|
*
|
|
393
|
-
*
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
*
|
|
316
|
+
* What lives here is only what a request needs regardless of what comes back:
|
|
317
|
+
* where to send it, with what credentials, under what bounds, and with which retry
|
|
318
|
+
* policy. A pipeline's own vocabulary stays with the pipeline — the generation
|
|
319
|
+
* catalog is on {@link ../base/http-adapter.HttpConnection}, and the embedding
|
|
320
|
+
* catalog will be on its own extension, because merging the two catalogs is
|
|
321
|
+
* exactly the mistake that makes one model shape stand in for another.
|
|
397
322
|
*
|
|
398
|
-
*
|
|
399
|
-
* into one string — the wording classifiers need all three because providers
|
|
400
|
-
* disagree about which field carries the useful part.
|
|
401
|
-
* @param status - status of a non-2xx response.
|
|
402
|
-
* @param detail - provider error text, joined; empty string when the body was unparseable.
|
|
403
|
-
* @returns the normalized code.
|
|
323
|
+
* @module ai-agent-sdk/providers/transport/connection
|
|
404
324
|
*/
|
|
405
|
-
function httpErrorCode(status, detail = "") {
|
|
406
|
-
if (status === 401 || status === 403) return MODEL_ERROR_CODES.AUTH;
|
|
407
|
-
if (status === 413) return MODEL_ERROR_CODES.INVALID_REQUEST;
|
|
408
|
-
if (isQuotaExceededError(detail)) return QUOTA_EXCEEDED_CODE;
|
|
409
|
-
if (status === 429) return MODEL_ERROR_CODES.RATE_LIMIT;
|
|
410
|
-
if (status === 400 || status === 422) return isContextWindowExceededError(detail) ? CONTEXT_WINDOW_EXCEEDED_CODE : MODEL_ERROR_CODES.INVALID_REQUEST;
|
|
411
|
-
if (status === 404) return MODEL_ERROR_CODES.INVALID_REQUEST;
|
|
412
|
-
if (status >= 500) return MODEL_ERROR_CODES.SERVER;
|
|
413
|
-
return `HTTP_${status}`;
|
|
414
|
-
}
|
|
415
325
|
/**
|
|
416
|
-
*
|
|
326
|
+
* Merge the transport's own header layer beneath the snapshot's auth layer, once.
|
|
417
327
|
*
|
|
418
|
-
*
|
|
419
|
-
*
|
|
420
|
-
*
|
|
421
|
-
*
|
|
422
|
-
* @returns a positive finite delay, or `undefined` when absent or unusable.
|
|
423
|
-
*/
|
|
424
|
-
function retryAfterMs(value) {
|
|
425
|
-
if (value === null) return void 0;
|
|
426
|
-
const trimmed = value.trim();
|
|
427
|
-
if (/^\d+$/.test(trimmed)) {
|
|
428
|
-
const delay = Number(trimmed) * 1e3;
|
|
429
|
-
return Number.isFinite(delay) && delay > 0 ? delay : void 0;
|
|
430
|
-
}
|
|
431
|
-
const delay = Date.parse(trimmed) - Date.now();
|
|
432
|
-
return Number.isFinite(delay) && delay > 0 ? delay : void 0;
|
|
433
|
-
}
|
|
434
|
-
/** Header names providers use for their request correlation id, in priority order. */
|
|
435
|
-
const REQUEST_ID_HEADERS = [
|
|
436
|
-
"request-id",
|
|
437
|
-
"x-request-id",
|
|
438
|
-
"x-requestid",
|
|
439
|
-
"cf-ray"
|
|
440
|
-
];
|
|
441
|
-
/**
|
|
442
|
-
* Extract a provider request id for diagnostics.
|
|
328
|
+
* Layer ownership is what makes this safe to call on a snapshot a subclass or a
|
|
329
|
+
* configuration produced: a transport header can never silently overwrite a
|
|
330
|
+
* credential, and the names the auth layer marked sensitive survive the merge so
|
|
331
|
+
* redaction still covers them.
|
|
443
332
|
*
|
|
444
|
-
*
|
|
445
|
-
*
|
|
446
|
-
*
|
|
447
|
-
* @
|
|
333
|
+
* A snapshot that already carries every layer — which configured adapters return —
|
|
334
|
+
* passes an empty transport layer and is returned untouched, so capturing twice
|
|
335
|
+
* cannot re-apply attribution.
|
|
336
|
+
* @param connection - the snapshot captured for this operation.
|
|
337
|
+
* @param transportHeaders - the transport layer to merge underneath; may be empty.
|
|
338
|
+
* @returns the snapshot with merged headers and the union of sensitive names.
|
|
448
339
|
*/
|
|
449
|
-
function
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
340
|
+
function captureTransportConnection(connection, transportHeaders) {
|
|
341
|
+
if (Reflect.ownKeys(transportHeaders).length === 0) return connection;
|
|
342
|
+
const merged = mergeHeaderLayers([
|
|
343
|
+
{
|
|
344
|
+
layer: "transport",
|
|
345
|
+
headers: transportHeaders
|
|
346
|
+
},
|
|
347
|
+
{
|
|
348
|
+
layer: "sdk-attribution",
|
|
349
|
+
headers: attributionHeaders()
|
|
350
|
+
},
|
|
351
|
+
{
|
|
352
|
+
layer: "auth",
|
|
353
|
+
headers: connection.headers
|
|
354
|
+
}
|
|
355
|
+
]);
|
|
356
|
+
return Object.freeze({
|
|
357
|
+
...connection,
|
|
358
|
+
headers: merged.headers,
|
|
359
|
+
sensitiveHeaderNames: Object.freeze([.../* @__PURE__ */ new Set([...connection.sensitiveHeaderNames ?? [], ...merged.sensitiveHeaderNames])])
|
|
360
|
+
});
|
|
454
361
|
}
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
362
|
+
|
|
363
|
+
//#endregion
|
|
364
|
+
//#region src/transport/limits.ts
|
|
365
|
+
/** Default end-to-end bound once provider request construction begins. */
|
|
366
|
+
const DEFAULT_REQUEST_TIMEOUT_MS = 6e5;
|
|
367
|
+
/** Default serialized request ceiling. */
|
|
368
|
+
const DEFAULT_MAX_REQUEST_BYTES = 33554432;
|
|
369
|
+
/** Default cumulative successful response-body ceiling. */
|
|
370
|
+
const DEFAULT_MAX_RESPONSE_BYTES = 33554432;
|
|
371
|
+
/** Default number of raw response chunks accepted from one request. */
|
|
372
|
+
const DEFAULT_MAX_RESPONSE_CHUNKS = 1e5;
|
|
373
|
+
/** Default error body retained for classification and diagnostics. */
|
|
374
|
+
const DEFAULT_MAX_ERROR_BODY_BYTES = 1048576;
|
|
375
|
+
/** Default diagnostic observer deadline; logging must never gate dispatch indefinitely. */
|
|
376
|
+
const DEFAULT_REQUEST_LOGGER_TIMEOUT_MS = 5e3;
|
|
377
|
+
function positiveInteger(value, name) {
|
|
378
|
+
if (!Number.isSafeInteger(value) || value < 1) throw new RangeError(`${name} must be a positive safe integer`);
|
|
379
|
+
return value;
|
|
380
|
+
}
|
|
381
|
+
function positiveFinite$1(value, name) {
|
|
382
|
+
if (!Number.isFinite(value) || value <= 0) throw new RangeError(`${name} must be a positive finite number`);
|
|
383
|
+
return value;
|
|
460
384
|
}
|
|
461
385
|
/**
|
|
462
|
-
*
|
|
386
|
+
* Apply defaults and reject nonsense bounds before any I/O happens.
|
|
463
387
|
*
|
|
464
|
-
*
|
|
465
|
-
*
|
|
466
|
-
*
|
|
467
|
-
* @param
|
|
468
|
-
* @returns
|
|
388
|
+
* Validation order is fixed and part of the contract: a configuration with two
|
|
389
|
+
* invalid bounds always reports the earlier field, so the error a caller sees does
|
|
390
|
+
* not depend on which limit the pipeline happens to consult first.
|
|
391
|
+
* @param connection - the captured snapshot this request is bound to.
|
|
392
|
+
* @returns fully resolved bounds; never partially defaulted.
|
|
469
393
|
*/
|
|
470
|
-
function
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
}
|
|
480
|
-
const error = typeof parsed === "object" && parsed !== null && "error" in parsed ? parsed.error : parsed;
|
|
481
|
-
const code = stringField(error, "code");
|
|
482
|
-
const type = stringField(error, "type");
|
|
483
|
-
const message = stringField(error, "message");
|
|
484
|
-
const detailField = stringField(error, "detail") ?? stringField(parsed, "detail");
|
|
485
|
-
const parts = [
|
|
486
|
-
code,
|
|
487
|
-
type,
|
|
488
|
-
message ?? detailField
|
|
489
|
-
].filter((part) => part !== void 0);
|
|
490
|
-
return {
|
|
491
|
-
message: message ?? detailField,
|
|
492
|
-
detail: parts.join(" ")
|
|
493
|
-
};
|
|
394
|
+
function resolveTransportLimits(connection) {
|
|
395
|
+
return Object.freeze({
|
|
396
|
+
requestTimeoutMs: positiveFinite$1(connection.requestTimeoutMs ?? 6e5, "requestTimeoutMs"),
|
|
397
|
+
maxRequestBytes: positiveInteger(connection.maxRequestBytes ?? 33554432, "maxRequestBytes"),
|
|
398
|
+
maxResponseBytes: positiveInteger(connection.maxResponseBytes ?? 33554432, "maxResponseBytes"),
|
|
399
|
+
maxResponseChunks: positiveInteger(connection.maxResponseChunks ?? 1e5, "maxResponseChunks"),
|
|
400
|
+
maxErrorBodyBytes: positiveInteger(connection.maxErrorBodyBytes ?? 1048576, "maxErrorBodyBytes"),
|
|
401
|
+
requestLoggerTimeoutMs: positiveFinite$1(connection.requestLoggerTimeoutMs ?? 5e3, "requestLoggerTimeoutMs")
|
|
402
|
+
});
|
|
494
403
|
}
|
|
495
404
|
|
|
496
405
|
//#endregion
|
|
497
|
-
//#region src/
|
|
406
|
+
//#region src/transport/http.ts
|
|
407
|
+
/**
|
|
408
|
+
* The pipeline-independent HTTP primitives the shared chain is built from.
|
|
409
|
+
*
|
|
410
|
+
* Each one exists because the obvious version of it is wrong in a way that only
|
|
411
|
+
* shows up under failure: a body read without a byte bound is a memory bug waiting
|
|
412
|
+
* for a hostile response, a promise raced against a signal that disposes nothing
|
|
413
|
+
* leaks a socket, a redirect followed once is a credential sent somewhere the
|
|
414
|
+
* caller never named, and an iterator abandoned without `return()` leaves a reader
|
|
415
|
+
* locked forever.
|
|
416
|
+
*
|
|
417
|
+
* None of it knows anything about SSE, JSON, catalogs, or model shapes — which is
|
|
418
|
+
* why it sits under the transport and not beside a pipeline.
|
|
419
|
+
*
|
|
420
|
+
* @module ai-agent-sdk/providers/transport/http
|
|
421
|
+
*/
|
|
422
|
+
/** Bound a response body by cumulative bytes and by chunk count, whichever trips first. */
|
|
498
423
|
function boundedResponseBody(source, maxBytes, maxChunks, displayName, signal) {
|
|
499
424
|
let bytes = 0;
|
|
500
425
|
let chunks = 0;
|
|
@@ -506,6 +431,7 @@ function boundedResponseBody(source, maxBytes, maxChunks, displayName, signal) {
|
|
|
506
431
|
controller.enqueue(chunk);
|
|
507
432
|
} }), signal === void 0 ? void 0 : { signal });
|
|
508
433
|
}
|
|
434
|
+
/** Read at most `maxBytes` of text, marking the truncation in the returned string. */
|
|
509
435
|
async function readBoundedText(response, maxBytes, signal) {
|
|
510
436
|
if (response.body === null) return "";
|
|
511
437
|
const reader = response.body.getReader();
|
|
@@ -543,6 +469,7 @@ async function rejectProviderRedirect(response, requestedUrl) {
|
|
|
543
469
|
if (response.body !== null) await waitForSettlement(response.body.cancel().catch(() => void 0), 3e4);
|
|
544
470
|
throw new ModelError("provider transport rejected a redirect before following it", HTTP_PROVIDER_ERROR_CODES.REDIRECT_REJECTED, response.status === 0 ? void 0 : { status: response.status });
|
|
545
471
|
}
|
|
472
|
+
/** Await a promise but surrender as soon as the signal aborts. */
|
|
546
473
|
async function raceWithSignal(pending, signal) {
|
|
547
474
|
if (signal.aborted) {
|
|
548
475
|
pending.catch(() => void 0);
|
|
@@ -570,6 +497,7 @@ async function cancelResponseBody(response) {
|
|
|
570
497
|
await waitForSettlement(Promise.resolve().then(() => response.body.cancel()), 3e4);
|
|
571
498
|
} catch {}
|
|
572
499
|
}
|
|
500
|
+
/** Iterate an async source under a signal, closing the iterator when it is cut short. */
|
|
573
501
|
async function* withAbortSignal(iterable, signal) {
|
|
574
502
|
const iterator = iterable[Symbol.asyncIterator]();
|
|
575
503
|
let exhausted = false;
|
|
@@ -594,49 +522,501 @@ async function* withAbortSignal(iterable, signal) {
|
|
|
594
522
|
}
|
|
595
523
|
}
|
|
596
524
|
}
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
525
|
+
/** Join a base URL and a path while refusing credentials, cleartext, and origin escapes. */
|
|
526
|
+
function endpointUrl(baseUrl, path, allowInsecureHttp) {
|
|
527
|
+
let base;
|
|
528
|
+
try {
|
|
529
|
+
base = new URL(baseUrl);
|
|
530
|
+
} catch (error) {
|
|
531
|
+
throw new ModelError("provider baseUrl is not a valid absolute URL", MODEL_ERROR_CODES.INVALID_REQUEST, { cause: error });
|
|
532
|
+
}
|
|
533
|
+
if (base.username.length > 0 || base.password.length > 0) throw new ModelError("provider baseUrl must not contain credentials", MODEL_ERROR_CODES.INVALID_REQUEST);
|
|
534
|
+
if (base.search.length > 0 || base.hash.length > 0) throw new ModelError("provider baseUrl must not contain a query or fragment", MODEL_ERROR_CODES.INVALID_REQUEST);
|
|
535
|
+
if (base.protocol !== "https:" && !(allowInsecureHttp && base.protocol === "http:")) throw new ModelError("provider baseUrl must use HTTPS unless allowInsecureHttp is explicitly enabled", MODEL_ERROR_CODES.INVALID_REQUEST);
|
|
536
|
+
const normalizedBase = base.href.replace(/\/+$/, "");
|
|
537
|
+
let endpoint;
|
|
538
|
+
try {
|
|
539
|
+
endpoint = new URL(`${normalizedBase}${path}`);
|
|
540
|
+
} catch (error) {
|
|
541
|
+
throw new ModelError("provider endpoint path produced an invalid URL", MODEL_ERROR_CODES.INVALID_REQUEST, { cause: error });
|
|
542
|
+
}
|
|
543
|
+
if (endpoint.origin !== base.origin) throw new ModelError("provider endpoint path must remain on the configured origin", MODEL_ERROR_CODES.INVALID_REQUEST);
|
|
544
|
+
return endpoint;
|
|
545
|
+
}
|
|
546
|
+
/** Reduce a failure to what a support report may carry: a code, a status, no prose. */
|
|
547
|
+
function safeProviderFailure(failure) {
|
|
548
|
+
return Object.freeze({
|
|
549
|
+
type: "ModelError",
|
|
550
|
+
message: "provider attempt failed; inspect the stable code and request ID",
|
|
551
|
+
code: failure.code,
|
|
552
|
+
...failure.status === void 0 ? {} : { status: failure.status }
|
|
553
|
+
});
|
|
554
|
+
}
|
|
555
|
+
/** Replace credential-bearing header values, by provenance and by name shape. */
|
|
556
|
+
function redactHeaders(headers, sensitiveHeaderNames = []) {
|
|
557
|
+
const provenance = new Set(sensitiveHeaderNames.map((name) => name.toLowerCase()));
|
|
558
|
+
return Object.fromEntries(Object.entries(headers).map(([name, value]) => [name, provenance.has(name.toLowerCase()) || isSensitiveHeaderName(name) ? "[REDACTED]" : value]));
|
|
559
|
+
}
|
|
560
|
+
/** Local correlation id for a diagnostic record, without requiring a crypto global. */
|
|
561
|
+
function requestLogId() {
|
|
562
|
+
return globalThis.crypto?.randomUUID?.() ?? `request-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 10)}`;
|
|
563
|
+
}
|
|
564
|
+
/** The one shape a caller-requested cancellation takes on the way out. */
|
|
565
|
+
function abortError(displayName, cause) {
|
|
566
|
+
return new ModelError(`${displayName} request aborted by caller`, MODEL_ERROR_CODES.ABORTED, { cause });
|
|
567
|
+
}
|
|
568
|
+
|
|
569
|
+
//#endregion
|
|
570
|
+
//#region src/common/failure.ts
|
|
571
|
+
const ENCODER$1 = new TextEncoder();
|
|
572
|
+
const INVALID_FIELD = Symbol("invalid failure field");
|
|
573
|
+
/** Read an own data property without invoking getters or inherited state. */
|
|
574
|
+
function ownDataProbe(source, key) {
|
|
575
|
+
try {
|
|
576
|
+
const descriptor = Object.getOwnPropertyDescriptor(source, key);
|
|
577
|
+
if (descriptor === void 0) return {
|
|
578
|
+
present: false,
|
|
579
|
+
data: true
|
|
580
|
+
};
|
|
581
|
+
if (!("value" in descriptor)) return {
|
|
582
|
+
present: true,
|
|
583
|
+
data: false
|
|
584
|
+
};
|
|
585
|
+
return {
|
|
586
|
+
present: true,
|
|
587
|
+
data: true,
|
|
588
|
+
value: descriptor.value
|
|
589
|
+
};
|
|
590
|
+
} catch {
|
|
591
|
+
return {
|
|
592
|
+
present: true,
|
|
593
|
+
data: false
|
|
594
|
+
};
|
|
595
|
+
}
|
|
596
|
+
}
|
|
597
|
+
function boundedString(value, maxBytes) {
|
|
598
|
+
return typeof value === "string" && value.length > 0 && ENCODER$1.encode(value).byteLength <= maxBytes;
|
|
599
|
+
}
|
|
600
|
+
function optionalFailureField(source, key) {
|
|
601
|
+
const field = ownDataProbe(source, key);
|
|
602
|
+
return field.data ? field.value : INVALID_FIELD;
|
|
603
|
+
}
|
|
604
|
+
/**
|
|
605
|
+
* Validate the data twin carried by a ModelError from another core copy/realm.
|
|
606
|
+
* A lone outer code is deliberately insufficient: retry policy may trust a code
|
|
607
|
+
* only when the bounded inner envelope exists and agrees with it.
|
|
608
|
+
*/
|
|
609
|
+
function probeFailureEnvelope(value) {
|
|
610
|
+
if (typeof value !== "object" && typeof value !== "function" || value === null) return { kind: "absent" };
|
|
611
|
+
const outerCode = ownDataProbe(value, "code");
|
|
612
|
+
const carried = ownDataProbe(value, "failure");
|
|
613
|
+
if (!outerCode.present && !carried.present) return { kind: "absent" };
|
|
614
|
+
if (!outerCode.data || !carried.data || !boundedString(outerCode.value, HTTP_FOREIGN_FAILURE_LIMITS.codeBytes) || typeof carried.value !== "object" || carried.value === null || Array.isArray(carried.value)) return { kind: "invalid" };
|
|
615
|
+
const message = optionalFailureField(carried.value, "message");
|
|
616
|
+
const code = optionalFailureField(carried.value, "code");
|
|
617
|
+
const status = optionalFailureField(carried.value, "status");
|
|
618
|
+
const providerRetryAfterMs = optionalFailureField(carried.value, "providerRetryAfterMs");
|
|
619
|
+
const requestId = optionalFailureField(carried.value, "requestId");
|
|
620
|
+
if (message === INVALID_FIELD || code === INVALID_FIELD || status === INVALID_FIELD || providerRetryAfterMs === INVALID_FIELD || requestId === INVALID_FIELD || !boundedString(message, HTTP_FOREIGN_FAILURE_LIMITS.messageBytes) || !boundedString(code, HTTP_FOREIGN_FAILURE_LIMITS.codeBytes) || code !== outerCode.value || status !== void 0 && (!Number.isSafeInteger(status) || status < 100 || status > 599) || providerRetryAfterMs !== void 0 && (!Number.isFinite(providerRetryAfterMs) || providerRetryAfterMs <= 0) || requestId !== void 0 && !boundedString(requestId, HTTP_FOREIGN_FAILURE_LIMITS.requestIdBytes)) return { kind: "invalid" };
|
|
621
|
+
return {
|
|
622
|
+
kind: "valid",
|
|
623
|
+
failure: Object.freeze({
|
|
624
|
+
message,
|
|
625
|
+
code,
|
|
626
|
+
...status === void 0 ? {} : { status },
|
|
627
|
+
...providerRetryAfterMs === void 0 ? {} : { providerRetryAfterMs },
|
|
628
|
+
...requestId === void 0 ? {} : { requestId }
|
|
629
|
+
})
|
|
630
|
+
};
|
|
631
|
+
}
|
|
632
|
+
/** Normalize a foreign provider failure without relying on package class identity. */
|
|
633
|
+
function normalizeHttpBoundaryError(value, fallbackMessage) {
|
|
634
|
+
const envelope = probeFailureEnvelope(value);
|
|
635
|
+
if (envelope.kind === "absent") return new ModelError(fallbackMessage, MODEL_ERROR_CODES.TRANSPORT, { cause: value });
|
|
636
|
+
if (envelope.kind === "invalid") return new ModelError("provider supplied an invalid failure envelope", MODEL_ERROR_CODES.UNKNOWN, { cause: value });
|
|
637
|
+
const failure = envelope.failure;
|
|
638
|
+
return new ModelError(failure.message, failure.code, {
|
|
639
|
+
cause: value,
|
|
640
|
+
...failure.status === void 0 ? {} : { status: failure.status },
|
|
641
|
+
...failure.providerRetryAfterMs === void 0 ? {} : { providerRetryAfterMs: failure.providerRetryAfterMs },
|
|
642
|
+
...failure.requestId === void 0 ? {} : { requestId: failure.requestId }
|
|
643
|
+
});
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
//#endregion
|
|
647
|
+
//#region src/transport/errors.ts
|
|
648
|
+
/**
|
|
649
|
+
* The HTTP-to-taxonomy mapping every provider shares.
|
|
650
|
+
*
|
|
651
|
+
* Kept here rather than per provider because the interesting decisions are
|
|
652
|
+
* genuinely vendor-independent: a 429 that means "slow down" versus one that
|
|
653
|
+
* means "your balance is gone", and a 400 that means "your prompt is too long"
|
|
654
|
+
* versus one that means "your schema is wrong". Both distinctions are invisible
|
|
655
|
+
* in the status code and both change what the caller should do, so getting them
|
|
656
|
+
* right once is worth more than getting them right three times.
|
|
657
|
+
*
|
|
658
|
+
* It lives in the transport rather than beside the generation pipeline because
|
|
659
|
+
* every pipeline maps the same statuses the same way; the SSE half of generation
|
|
660
|
+
* has no opinion about what a 429 means.
|
|
661
|
+
*
|
|
662
|
+
* @module ai-agent-sdk/providers/transport/errors
|
|
663
|
+
*/
|
|
664
|
+
/**
|
|
665
|
+
* Map an HTTP status plus whatever the provider said into a stable code.
|
|
666
|
+
*
|
|
667
|
+
* `detail` should be the provider's error `code`, `type`, and `message` joined
|
|
668
|
+
* into one string — the wording classifiers need all three because providers
|
|
669
|
+
* disagree about which field carries the useful part.
|
|
670
|
+
* @param status - status of a non-2xx response.
|
|
671
|
+
* @param detail - provider error text, joined; empty string when the body was unparseable.
|
|
672
|
+
* @returns the normalized code.
|
|
673
|
+
*/
|
|
674
|
+
function httpErrorCode(status, detail = "") {
|
|
675
|
+
if (status === 401 || status === 403) return MODEL_ERROR_CODES.AUTH;
|
|
676
|
+
if (status === 413) return MODEL_ERROR_CODES.INVALID_REQUEST;
|
|
677
|
+
if (isQuotaExceededError(detail)) return QUOTA_EXCEEDED_CODE;
|
|
678
|
+
if (status === 429) return MODEL_ERROR_CODES.RATE_LIMIT;
|
|
679
|
+
if (status === 400 || status === 422) return isContextWindowExceededError(detail) ? CONTEXT_WINDOW_EXCEEDED_CODE : MODEL_ERROR_CODES.INVALID_REQUEST;
|
|
680
|
+
if (status === 404) return MODEL_ERROR_CODES.INVALID_REQUEST;
|
|
681
|
+
if (status >= 500) return MODEL_ERROR_CODES.SERVER;
|
|
682
|
+
return `HTTP_${status}`;
|
|
683
|
+
}
|
|
684
|
+
/**
|
|
685
|
+
* Parse a `retry-after` header into milliseconds.
|
|
686
|
+
*
|
|
687
|
+
* The header comes in two forms — delta-seconds and an HTTP date — and both are
|
|
688
|
+
* used in practice. A date already in the past yields `undefined` rather than a
|
|
689
|
+
* negative delay.
|
|
690
|
+
* @param value - the raw header value, or `null` when absent.
|
|
691
|
+
* @returns a positive finite delay, or `undefined` when absent or unusable.
|
|
692
|
+
*/
|
|
693
|
+
function retryAfterMs(value) {
|
|
694
|
+
if (value === null) return void 0;
|
|
695
|
+
const trimmed = value.trim();
|
|
696
|
+
if (/^\d+$/.test(trimmed)) {
|
|
697
|
+
const delay = Number(trimmed) * 1e3;
|
|
698
|
+
return Number.isFinite(delay) && delay > 0 ? delay : void 0;
|
|
699
|
+
}
|
|
700
|
+
const delay = Date.parse(trimmed) - Date.now();
|
|
701
|
+
return Number.isFinite(delay) && delay > 0 ? delay : void 0;
|
|
702
|
+
}
|
|
703
|
+
/** Header names providers use for their request correlation id, in priority order. */
|
|
704
|
+
const REQUEST_ID_HEADERS = [
|
|
705
|
+
"request-id",
|
|
706
|
+
"x-request-id",
|
|
707
|
+
"x-requestid",
|
|
708
|
+
"cf-ray"
|
|
709
|
+
];
|
|
710
|
+
/**
|
|
711
|
+
* Extract a provider request id for diagnostics.
|
|
712
|
+
*
|
|
713
|
+
* Worth capturing even though nothing programmatic reads it: when a provider is
|
|
714
|
+
* misbehaving, this id is what their support needs to find the request.
|
|
715
|
+
* @param headers - the response headers.
|
|
716
|
+
* @returns the first non-empty id found, or `undefined`.
|
|
717
|
+
*/
|
|
718
|
+
function requestIdFrom(headers) {
|
|
719
|
+
for (const name of REQUEST_ID_HEADERS) {
|
|
720
|
+
const value = headers.get(name);
|
|
721
|
+
if (value !== null && value.length > 0) return ProviderRequestId(value);
|
|
722
|
+
}
|
|
723
|
+
}
|
|
724
|
+
/** Read a string property from an unknown object without trusting its shape. */
|
|
725
|
+
function stringField(source, key) {
|
|
726
|
+
if (typeof source !== "object" || source === null) return void 0;
|
|
727
|
+
const value = source[key];
|
|
728
|
+
return typeof value === "string" && value.length > 0 ? value : void 0;
|
|
604
729
|
}
|
|
605
|
-
|
|
606
|
-
|
|
730
|
+
/**
|
|
731
|
+
* Reduce a provider error body to a message and a classifier detail string.
|
|
732
|
+
*
|
|
733
|
+
* Handles the two shapes both providers use — `{error: {...}}` and a bare
|
|
734
|
+
* `{type, message}` — and tolerates a body that is not JSON at all, which is what
|
|
735
|
+
* a gateway or load balancer in front of the provider will return.
|
|
736
|
+
* @param raw - the response body as text.
|
|
737
|
+
* @returns the message and joined detail.
|
|
738
|
+
*/
|
|
739
|
+
function parseErrorBody(raw) {
|
|
740
|
+
let parsed;
|
|
607
741
|
try {
|
|
608
|
-
|
|
609
|
-
} catch
|
|
610
|
-
|
|
742
|
+
parsed = JSON.parse(raw);
|
|
743
|
+
} catch {
|
|
744
|
+
return {
|
|
745
|
+
message: void 0,
|
|
746
|
+
detail: raw.slice(0, 2048)
|
|
747
|
+
};
|
|
611
748
|
}
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
const
|
|
616
|
-
|
|
749
|
+
const error = typeof parsed === "object" && parsed !== null && "error" in parsed ? parsed.error : parsed;
|
|
750
|
+
const code = stringField(error, "code");
|
|
751
|
+
const type = stringField(error, "type");
|
|
752
|
+
const message = stringField(error, "message");
|
|
753
|
+
const detailField = stringField(error, "detail") ?? stringField(parsed, "detail");
|
|
754
|
+
const parts = [
|
|
755
|
+
code,
|
|
756
|
+
type,
|
|
757
|
+
message ?? detailField
|
|
758
|
+
].filter((part) => part !== void 0);
|
|
759
|
+
return {
|
|
760
|
+
message: message ?? detailField,
|
|
761
|
+
detail: parts.join(" ")
|
|
762
|
+
};
|
|
763
|
+
}
|
|
764
|
+
|
|
765
|
+
//#endregion
|
|
766
|
+
//#region src/transport/session.ts
|
|
767
|
+
/**
|
|
768
|
+
* The one risky chain every HTTP pipeline in this package runs, written once.
|
|
769
|
+
*
|
|
770
|
+
* Everything here is a step that is invisible when it works and expensive when it
|
|
771
|
+
* is missing: fusing the caller's cancellation with our own teardown controller and
|
|
772
|
+
* the request deadline, bounding the outbound body, letting a diagnostic observer
|
|
773
|
+
* look at the request without letting it veto dispatch, opening a provider attempt
|
|
774
|
+
* before the socket and closing it exactly once afterwards, refusing a redirect
|
|
775
|
+
* instead of replaying credentials to wherever it points, turning a non-2xx into a
|
|
776
|
+
* stable code with `retry-after` and a request id attached, and releasing the
|
|
777
|
+
* response body when the consumer walks away early.
|
|
778
|
+
*
|
|
779
|
+
* A second copy of this chain for embedding would be a second chance to forget one
|
|
780
|
+
* of those steps — which is precisely why the chain, and not the decoding, is what
|
|
781
|
+
* gets shared. What a pipeline supplies is only `decode`: what to do with a
|
|
782
|
+
* response that already passed every guard above.
|
|
783
|
+
*
|
|
784
|
+
* The classification order at the bottom is part of the contract and is deliberately
|
|
785
|
+
* not simplified: a fired deadline outranks an abort the caller did not request,
|
|
786
|
+
* an aborted fused signal outranks a transport failure, and an admission refusal
|
|
787
|
+
* from `startProviderAttempt` is rethrown untouched so audit mode's decision is not
|
|
788
|
+
* relabelled as a transport error.
|
|
789
|
+
*
|
|
790
|
+
* @module ai-agent-sdk/providers/transport/session
|
|
791
|
+
*/
|
|
792
|
+
/**
|
|
793
|
+
* Run one request through the shared safety chain and stream `use`'s output.
|
|
794
|
+
*
|
|
795
|
+
* The generator shape matters: the provider attempt stays open, and the response
|
|
796
|
+
* body stays owned, for as long as the consumer keeps pulling. A consumer that
|
|
797
|
+
* stops early aborts the teardown controller in `finally`, which is what tears down
|
|
798
|
+
* an in-flight response instead of leaking the connection.
|
|
799
|
+
* @param input - the request facts, all captured from one connection snapshot.
|
|
800
|
+
* @param use - decodes a guarded response; its failures are classified here.
|
|
801
|
+
* @returns whatever `use` yields, unchanged.
|
|
802
|
+
*/
|
|
803
|
+
async function* withTransportSession(input, use) {
|
|
804
|
+
const { connection, context, displayName, model, provider } = input;
|
|
805
|
+
const callerSignal = input.signal;
|
|
806
|
+
const consumer = new AbortController();
|
|
807
|
+
const limits = resolveTransportLimits(connection);
|
|
808
|
+
const timeout = AbortSignal.timeout(limits.requestTimeoutMs);
|
|
809
|
+
const signal = AbortSignal.any([
|
|
810
|
+
consumer.signal,
|
|
811
|
+
timeout,
|
|
812
|
+
...callerSignal === void 0 ? [] : [callerSignal]
|
|
813
|
+
]);
|
|
814
|
+
let admissionFailure;
|
|
815
|
+
let ownedResponse;
|
|
617
816
|
try {
|
|
618
|
-
|
|
817
|
+
signal.throwIfAborted();
|
|
818
|
+
const preparedBody = typeof input.body === "function" ? await input.body(signal) : input.body;
|
|
819
|
+
if (preparedBody.bytes > limits.maxRequestBytes) throw new ModelError(`${displayName} request exceeds the ${limits.maxRequestBytes}-byte limit`, MODEL_ERROR_CODES.INVALID_REQUEST);
|
|
820
|
+
const endpoint = endpointUrl(connection.baseUrl, input.path, connection.allowInsecureHttp ?? false);
|
|
821
|
+
const url = endpoint.href;
|
|
822
|
+
const origin = endpoint.origin;
|
|
823
|
+
const headers = connection.headers;
|
|
824
|
+
try {
|
|
825
|
+
const loggerSignal = AbortSignal.any([signal, AbortSignal.timeout(limits.requestLoggerTimeoutMs)]);
|
|
826
|
+
await raceWithSignal(Promise.resolve(input.observeRequest?.({
|
|
827
|
+
schemaVersion: 1,
|
|
828
|
+
type: "provider-request",
|
|
829
|
+
id: requestLogId(),
|
|
830
|
+
timestamp: (/* @__PURE__ */ new Date()).toISOString(),
|
|
831
|
+
provider,
|
|
832
|
+
model,
|
|
833
|
+
method: "POST",
|
|
834
|
+
url,
|
|
835
|
+
headers: redactHeaders(headers, connection.sensitiveHeaderNames),
|
|
836
|
+
body: preparedBody.value,
|
|
837
|
+
bodyBytes: preparedBody.bytes
|
|
838
|
+
})), loggerSignal);
|
|
839
|
+
} catch {}
|
|
840
|
+
let attempt;
|
|
841
|
+
let dispatchState = "not-sent";
|
|
842
|
+
let httpStatus;
|
|
843
|
+
let providerRequestId;
|
|
844
|
+
let attemptStatus = "unknown";
|
|
845
|
+
let attemptUsage;
|
|
846
|
+
let usageFinal = true;
|
|
847
|
+
let attemptError;
|
|
848
|
+
try {
|
|
849
|
+
signal.throwIfAborted();
|
|
850
|
+
try {
|
|
851
|
+
attempt = await context?.startProviderAttempt?.({
|
|
852
|
+
provider,
|
|
853
|
+
model,
|
|
854
|
+
method: "POST",
|
|
855
|
+
origin
|
|
856
|
+
}, signal);
|
|
857
|
+
} catch (error) {
|
|
858
|
+
admissionFailure = { value: error };
|
|
859
|
+
throw error;
|
|
860
|
+
}
|
|
861
|
+
signal.throwIfAborted();
|
|
862
|
+
dispatchState = "unknown";
|
|
863
|
+
const pendingResponse = (connection.fetch ?? globalThis.fetch)(url, {
|
|
864
|
+
method: "POST",
|
|
865
|
+
headers,
|
|
866
|
+
body: preparedBody.encoded,
|
|
867
|
+
signal,
|
|
868
|
+
redirect: "manual"
|
|
869
|
+
});
|
|
870
|
+
pendingResponse.then((response) => {
|
|
871
|
+
if (signal.aborted) return cancelResponseBody(response);
|
|
872
|
+
}, () => void 0);
|
|
873
|
+
const response = await raceWithSignal(pendingResponse, signal);
|
|
874
|
+
ownedResponse = response;
|
|
875
|
+
signal.throwIfAborted();
|
|
876
|
+
dispatchState = "sent";
|
|
877
|
+
httpStatus = response.status;
|
|
878
|
+
providerRequestId = requestIdFrom(response.headers);
|
|
879
|
+
await rejectProviderRedirect(response, url);
|
|
880
|
+
if (!response.ok) throw await httpFailure(response, origin, limits.maxErrorBodyBytes, signal, input);
|
|
881
|
+
yield* use(Object.freeze({
|
|
882
|
+
response,
|
|
883
|
+
url,
|
|
884
|
+
origin,
|
|
885
|
+
signal,
|
|
886
|
+
accept: input.accept,
|
|
887
|
+
limits,
|
|
888
|
+
...providerRequestId === void 0 ? {} : { providerRequestId },
|
|
889
|
+
reportUsage(usage, final = true) {
|
|
890
|
+
attemptUsage = usage;
|
|
891
|
+
usageFinal = final;
|
|
892
|
+
},
|
|
893
|
+
...attempt === void 0 ? {} : { attemptId: attempt.attemptId },
|
|
894
|
+
reportOutcome(status, failure) {
|
|
895
|
+
attemptStatus = status;
|
|
896
|
+
if (failure !== void 0) attemptError = safeProviderFailure(failure);
|
|
897
|
+
}
|
|
898
|
+
}));
|
|
899
|
+
} catch (error) {
|
|
900
|
+
if (admissionFailure !== void 0 && error === admissionFailure.value) throw error;
|
|
901
|
+
const mapped = timeout.aborted && callerSignal?.aborted !== true ? new ModelError(`${displayName} request exceeded its ${limits.requestTimeoutMs}ms time limit`, MODEL_ERROR_CODES.TIMEOUT, { cause: error }) : signal.aborted ? abortError(displayName, error) : normalizeHttpBoundaryError(error, `${displayName} request to ${origin} failed`);
|
|
902
|
+
attemptStatus = mapped.code === MODEL_ERROR_CODES.ABORTED ? "aborted" : "error";
|
|
903
|
+
attemptError = safeProviderFailure(mapped.failure);
|
|
904
|
+
throw mapped;
|
|
905
|
+
} finally {
|
|
906
|
+
attempt?.end({
|
|
907
|
+
status: attemptStatus,
|
|
908
|
+
dispatchState,
|
|
909
|
+
...attemptUsage === void 0 ? {} : { reported: attemptUsage },
|
|
910
|
+
...usageFinal ? {} : { usageFinal: false },
|
|
911
|
+
...httpStatus === void 0 ? {} : { httpStatus },
|
|
912
|
+
...providerRequestId === void 0 ? {} : { providerRequestId },
|
|
913
|
+
...attemptError === void 0 ? {} : { error: attemptError }
|
|
914
|
+
});
|
|
915
|
+
}
|
|
619
916
|
} catch (error) {
|
|
620
|
-
|
|
917
|
+
if (callerSignal?.aborted === true) throw abortError(displayName, error);
|
|
918
|
+
if (timeout.aborted) throw new ModelError(`${displayName} request exceeded its ${limits.requestTimeoutMs}ms time limit`, MODEL_ERROR_CODES.TIMEOUT, { cause: error });
|
|
919
|
+
if (admissionFailure !== void 0 && error === admissionFailure.value) throw error;
|
|
920
|
+
throw normalizeHttpBoundaryError(error, `${displayName} stream failed`);
|
|
921
|
+
} finally {
|
|
922
|
+
consumer.abort(/* @__PURE__ */ new Error(`${displayName} stream consumer stopped`));
|
|
923
|
+
if (ownedResponse !== void 0) await cancelResponseBody(ownedResponse);
|
|
621
924
|
}
|
|
622
|
-
if (endpoint.origin !== base.origin) throw new ModelError("provider endpoint path must remain on the configured origin", MODEL_ERROR_CODES.INVALID_REQUEST);
|
|
623
|
-
return endpoint;
|
|
624
925
|
}
|
|
625
|
-
|
|
626
|
-
|
|
627
|
-
|
|
628
|
-
|
|
629
|
-
|
|
630
|
-
|
|
926
|
+
/**
|
|
927
|
+
* Turn a non-2xx response into a fully populated {@link ModelError}.
|
|
928
|
+
*
|
|
929
|
+
* The body is read under the same bound and the same signal as everything else,
|
|
930
|
+
* and a body that cannot be read does not replace the status — the status is the
|
|
931
|
+
* more reliable signal of the two anyway.
|
|
932
|
+
* @param response - the non-success response.
|
|
933
|
+
* @param origin - request origin, reported in the message instead of the full URL.
|
|
934
|
+
* @param maxBytes - error-body bound from the resolved limits.
|
|
935
|
+
* @param signal - the fused signal.
|
|
936
|
+
* @param input - request facts, for the display name and code override.
|
|
937
|
+
* @returns the mapped error, carrying status, `retry-after`, and request id.
|
|
938
|
+
*/
|
|
939
|
+
async function httpFailure(response, origin, maxBytes, signal, input) {
|
|
940
|
+
let raw = "";
|
|
941
|
+
try {
|
|
942
|
+
raw = await readBoundedText(response, maxBytes, signal);
|
|
943
|
+
} catch {}
|
|
944
|
+
const { message, detail } = parseErrorBody(raw);
|
|
945
|
+
const delay = retryAfterMs(response.headers.get("retry-after"));
|
|
946
|
+
const id = requestIdFrom(response.headers);
|
|
947
|
+
return new ModelError(message ?? `${input.displayName} error (HTTP ${response.status}) from ${origin}`, (input.errorCode ?? httpErrorCode)(response.status, detail), {
|
|
948
|
+
cause: new Error(raw.length > 0 ? raw : `HTTP ${response.status}`),
|
|
949
|
+
status: response.status,
|
|
950
|
+
...delay === void 0 ? {} : { providerRetryAfterMs: delay },
|
|
951
|
+
...id === void 0 ? {} : { requestId: id }
|
|
631
952
|
});
|
|
632
953
|
}
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
954
|
+
|
|
955
|
+
//#endregion
|
|
956
|
+
//#region src/transport/stream.ts
|
|
957
|
+
/**
|
|
958
|
+
* The streaming pipeline's entry into the shared transport chain.
|
|
959
|
+
*
|
|
960
|
+
* Two properties define it, and both are about what happens while a consumer is
|
|
961
|
+
* still pulling. The provider attempt stays open for the whole stream, because a
|
|
962
|
+
* stream that fails halfway through is one attempt with a failure — not a success
|
|
963
|
+
* followed by a mystery. And every pull is raced against the fused signal, so a
|
|
964
|
+
* decoder that blocks on a provider which has stopped sending still surrenders when
|
|
965
|
+
* the caller aborts, the deadline fires, or the consumer walks away.
|
|
966
|
+
*
|
|
967
|
+
* What `decode` sees is a response that already cleared every guard in
|
|
968
|
+
* {@link withTransportSession}: bounded body, no redirect, 2xx, attempt open. What
|
|
969
|
+
* it owns is the format — media type, framing, termination.
|
|
970
|
+
*
|
|
971
|
+
* @module ai-agent-sdk/providers/transport/stream
|
|
972
|
+
*/
|
|
973
|
+
/**
|
|
974
|
+
* Send one request and stream whatever `decode` makes of the response.
|
|
975
|
+
*
|
|
976
|
+
* @param input - request facts captured from one connection snapshot.
|
|
977
|
+
* @param decode - turns a guarded response into the pipeline's own values.
|
|
978
|
+
* @returns the decoded values, with the attempt held open until iteration ends.
|
|
979
|
+
*/
|
|
980
|
+
function transportStream(input, decode) {
|
|
981
|
+
return withTransportSession(input, (session) => withAbortSignal(decode(session), session.signal));
|
|
636
982
|
}
|
|
637
|
-
|
|
638
|
-
|
|
983
|
+
|
|
984
|
+
//#endregion
|
|
985
|
+
//#region src/base/context-policy.ts
|
|
986
|
+
function applyModelContextPolicy(info, policy, configured, providerOverride) {
|
|
987
|
+
const defaultContextWindow = configured?.defaultContextWindow ?? policy?.defaultContextWindow ?? info.context?.defaultContextWindow;
|
|
988
|
+
const knownMaximum = policy?.maxContextWindow ?? info.context?.maxContextWindow;
|
|
989
|
+
const maxContextWindow = knownMaximum === void 0 ? configured?.maxContextWindow : Math.min(knownMaximum, configured?.maxContextWindow ?? knownMaximum);
|
|
990
|
+
const standardPriceInputTokens = configured?.standardPriceInputTokens ?? policy?.standardPriceInputTokens ?? info.context?.standardPriceInputTokens;
|
|
991
|
+
const contextWindow = configured?.contextWindow ?? providerOverride ?? defaultContextWindow ?? info.context?.contextWindow;
|
|
992
|
+
for (const [name, value] of Object.entries({
|
|
993
|
+
contextWindow,
|
|
994
|
+
defaultContextWindow,
|
|
995
|
+
maxContextWindow,
|
|
996
|
+
standardPriceInputTokens
|
|
997
|
+
})) if (value !== void 0 && (!Number.isSafeInteger(value) || value <= 0)) throw new TypeError(`${name} must be a positive safe integer`);
|
|
998
|
+
if (maxContextWindow !== void 0 && (contextWindow !== void 0 && contextWindow > maxContextWindow || defaultContextWindow !== void 0 && defaultContextWindow > maxContextWindow)) throw new RangeError("contextWindow exceeds maxContextWindow");
|
|
999
|
+
if (contextWindow === void 0) return info;
|
|
1000
|
+
return {
|
|
1001
|
+
...info,
|
|
1002
|
+
context: {
|
|
1003
|
+
contextWindow,
|
|
1004
|
+
...defaultContextWindow === void 0 ? {} : { defaultContextWindow },
|
|
1005
|
+
...maxContextWindow === void 0 ? {} : { maxContextWindow },
|
|
1006
|
+
...standardPriceInputTokens === void 0 ? {} : { standardPriceInputTokens },
|
|
1007
|
+
...standardPriceInputTokens !== void 0 && contextWindow > standardPriceInputTokens ? { pricingWarning: "extended-context-may-cost-more" } : {}
|
|
1008
|
+
}
|
|
1009
|
+
};
|
|
639
1010
|
}
|
|
1011
|
+
/** Capture configuration so later caller mutations cannot change model resolution. */
|
|
1012
|
+
function createModelContextPolicy(policies, models, providerOverride) {
|
|
1013
|
+
const captured = structuredClone(policies);
|
|
1014
|
+
const configured = structuredClone(models ?? []);
|
|
1015
|
+
return (info) => applyModelContextPolicy(info, Object.hasOwn(captured, info.id) ? captured[info.id] : void 0, configured.find((model) => model.id === info.id), providerOverride);
|
|
1016
|
+
}
|
|
1017
|
+
|
|
1018
|
+
//#endregion
|
|
1019
|
+
//#region src/base/transport.ts
|
|
640
1020
|
function catalogModelInfo(provider, model) {
|
|
641
1021
|
return {
|
|
642
1022
|
provider,
|
|
@@ -651,7 +1031,7 @@ function catalogModelInfo(provider, model) {
|
|
|
651
1031
|
/** Resolve exact metadata from an advisory catalog without opening a connection. */
|
|
652
1032
|
function resolvedCatalogModelInfo(provider, modelId, models, defaultMaxTokens, defaultContextWindow) {
|
|
653
1033
|
const configured = models.find((entry) => entry.id === modelId);
|
|
654
|
-
return {
|
|
1034
|
+
return applyModelContextPolicy({
|
|
655
1035
|
...configured === void 0 ? {
|
|
656
1036
|
provider,
|
|
657
1037
|
id: modelId,
|
|
@@ -659,14 +1039,11 @@ function resolvedCatalogModelInfo(provider, modelId, models, defaultMaxTokens, d
|
|
|
659
1039
|
inputModalities: ["text"]
|
|
660
1040
|
} : catalogModelInfo(provider, configured),
|
|
661
1041
|
context: { contextWindow: configured?.contextWindow ?? defaultContextWindow },
|
|
662
|
-
defaultMaxTokens: configured?.maxTokens ?? defaultMaxTokens,
|
|
663
|
-
maxOutputTokens: configured
|
|
1042
|
+
defaultMaxTokens: configured?.defaultMaxTokens ?? configured?.maxTokens ?? defaultMaxTokens,
|
|
1043
|
+
...configured?.maxTokens === void 0 ? {} : { maxOutputTokens: configured.maxTokens },
|
|
664
1044
|
...configured?.reasoning === void 0 ? {} : { reasoning: configured.reasoning },
|
|
665
1045
|
...configured?.outputModalities === void 0 ? {} : { outputModalities: configured.outputModalities }
|
|
666
|
-
};
|
|
667
|
-
}
|
|
668
|
-
function abortError(displayName, cause) {
|
|
669
|
-
return new ModelError(`${displayName} request aborted by caller`, MODEL_ERROR_CODES.ABORTED, { cause });
|
|
1046
|
+
}, void 0, configured);
|
|
670
1047
|
}
|
|
671
1048
|
|
|
672
1049
|
//#endregion
|
|
@@ -689,22 +1066,33 @@ function abortError(displayName, cause) {
|
|
|
689
1066
|
* request, HTTP error mapping, `retry-after`, request ids, SSE decoding, the idle
|
|
690
1067
|
* bound, and teardown — is shared and happens exactly once, here.
|
|
691
1068
|
*
|
|
1069
|
+
* The risky half of that list — signal fusion, request bounds, the diagnostic
|
|
1070
|
+
* observer, attempt accounting, redirect refusal, status mapping, teardown — now
|
|
1071
|
+
* lives in {@link ../transport/session.withTransportSession} so a second pipeline
|
|
1072
|
+
* cannot reimplement it slightly differently. What stays in this file is what is
|
|
1073
|
+
* genuinely generation's: the modality guard, the catalog, the serialized-body
|
|
1074
|
+
* cache for one prepared call, and {@link HttpModelAdapter.decodeSse} — the SSE
|
|
1075
|
+
* half, unchanged, applied to a response that already cleared every guard.
|
|
1076
|
+
*
|
|
692
1077
|
* @module ai-agent-sdk/providers/base/http-adapter
|
|
693
1078
|
*/
|
|
694
1079
|
/** Default idle bound: five minutes without a single byte is a hung stream. */
|
|
695
1080
|
const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 3e5;
|
|
696
|
-
/**
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
1081
|
+
/**
|
|
1082
|
+
* Resolve the SSE bounds before any transport work starts.
|
|
1083
|
+
*
|
|
1084
|
+
* Deliberately validated in the pipeline and not inside `decodeSse`: an
|
|
1085
|
+
* unusable bound is a configuration error, and configuration errors must not
|
|
1086
|
+
* arrive after a provider attempt has been opened and a request sent.
|
|
1087
|
+
* @param connection - the snapshot this call is bound to.
|
|
1088
|
+
* @returns defaulted, validated event bounds.
|
|
1089
|
+
*/
|
|
1090
|
+
function resolveSseLimits(connection) {
|
|
1091
|
+
return Object.freeze({
|
|
1092
|
+
maxEvents: positiveInteger(connection.maxSseEvents ?? 1e5, "maxSseEvents"),
|
|
1093
|
+
maxEventChars: positiveInteger(connection.maxSseEventChars ?? 1048576, "maxSseEventChars")
|
|
1094
|
+
});
|
|
1095
|
+
}
|
|
708
1096
|
/** Base for every HTTP provider adapter in this package. */
|
|
709
1097
|
var HttpModelAdapter = class extends ModelAdapter {
|
|
710
1098
|
/**
|
|
@@ -781,204 +1169,115 @@ var HttpModelAdapter = class extends ModelAdapter {
|
|
|
781
1169
|
}
|
|
782
1170
|
/** Capture legacy subclass transport/auth layers once; configured adapters already return all five. */
|
|
783
1171
|
captureConnection(connection) {
|
|
784
|
-
|
|
785
|
-
if (Reflect.ownKeys(transport).length === 0) return connection;
|
|
786
|
-
const merged = mergeHeaderLayers([
|
|
787
|
-
{
|
|
788
|
-
layer: "transport",
|
|
789
|
-
headers: transport
|
|
790
|
-
},
|
|
791
|
-
{
|
|
792
|
-
layer: "sdk-attribution",
|
|
793
|
-
headers: attributionHeaders()
|
|
794
|
-
},
|
|
795
|
-
{
|
|
796
|
-
layer: "auth",
|
|
797
|
-
headers: connection.headers
|
|
798
|
-
}
|
|
799
|
-
]);
|
|
800
|
-
return Object.freeze({
|
|
801
|
-
...connection,
|
|
802
|
-
headers: merged.headers,
|
|
803
|
-
sensitiveHeaderNames: Object.freeze([.../* @__PURE__ */ new Set([...connection.sensitiveHeaderNames ?? [], ...merged.sensitiveHeaderNames])])
|
|
804
|
-
});
|
|
1172
|
+
return captureTransportConnection(connection, this.baseHeaders());
|
|
805
1173
|
}
|
|
806
1174
|
/**
|
|
807
|
-
* The
|
|
1175
|
+
* The generation pipeline: guard the modalities, then hand one request to the
|
|
1176
|
+
* shared transport chain with SSE decoding as its only pipeline-specific part.
|
|
1177
|
+
*
|
|
1178
|
+
* Still a generator, and deliberately so: the modality guard, the body
|
|
1179
|
+
* serialization and every transport step stay lazy until a consumer pulls, which
|
|
1180
|
+
* is the behaviour every existing caller of `stream()` already relies on.
|
|
808
1181
|
*/
|
|
809
1182
|
async *run(options, connection, model, context, wireBodyCache = {}) {
|
|
810
1183
|
context?.declareProviderAttemptAccounting?.();
|
|
811
1184
|
if (options.messages.some((message) => contentHasImage(message.content)) && model.inputModalities?.includes("image") !== true) throw new ModelError(`${this.displayName} model "${options.model}" does not accept image input`, MODEL_ERROR_CODES.UNSUPPORTED_CONTENT);
|
|
1185
|
+
if (options.messages.some((message) => contentHasDocument(message.content)) && model.inputModalities?.includes("document") !== true) throw new ModelError(`${this.displayName} model "${options.model}" does not accept document input`, MODEL_ERROR_CODES.UNSUPPORTED_CONTENT);
|
|
812
1186
|
const request = {
|
|
813
1187
|
options,
|
|
814
1188
|
model,
|
|
815
1189
|
connection,
|
|
816
1190
|
maxTokens: options.maxTokens ?? model.defaultMaxTokens ?? connection.defaultMaxTokens
|
|
817
1191
|
};
|
|
818
|
-
const
|
|
819
|
-
const
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
836
|
-
signal.
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
843
|
-
|
|
844
|
-
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
|
|
848
|
-
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
admissionFailure = { value: error };
|
|
879
|
-
throw error;
|
|
880
|
-
}
|
|
881
|
-
signal.throwIfAborted();
|
|
882
|
-
dispatchState = "unknown";
|
|
883
|
-
const pendingResponse = (connection.fetch ?? globalThis.fetch)(url, {
|
|
884
|
-
method: "POST",
|
|
885
|
-
headers,
|
|
886
|
-
body,
|
|
887
|
-
signal,
|
|
888
|
-
redirect: "manual"
|
|
889
|
-
});
|
|
890
|
-
pendingResponse.then((response) => {
|
|
891
|
-
if (signal.aborted) return cancelResponseBody(response);
|
|
892
|
-
}, () => void 0);
|
|
893
|
-
const response = await raceWithSignal(pendingResponse, signal);
|
|
894
|
-
ownedResponse = response;
|
|
895
|
-
signal.throwIfAborted();
|
|
896
|
-
dispatchState = "sent";
|
|
897
|
-
httpStatus = response.status;
|
|
898
|
-
providerRequestId = requestIdFrom(response.headers);
|
|
899
|
-
await rejectProviderRedirect(response, url);
|
|
900
|
-
if (!response.ok) throw await this.httpFailure(response, origin, maxErrorBodyBytes, signal);
|
|
901
|
-
if (response.headers.get("content-type")?.split(";", 1)[0]?.trim().toLowerCase() !== "text/event-stream") throw new ModelError(`${this.displayName} response is not text/event-stream`, HTTP_PROVIDER_ERROR_CODES.STREAM_MEDIA_TYPE_INVALID);
|
|
902
|
-
if (response.body === null) throw new ModelError(`${this.displayName} returned no response body`, MODEL_ERROR_CODES.STREAM_CLOSED);
|
|
903
|
-
const declaredLength = response.headers.get("content-length");
|
|
904
|
-
if (declaredLength !== null && /^\d+$/.test(declaredLength) && Number(declaredLength) > maxResponseBytes) throw new ModelError(`${this.displayName} response exceeds the ${maxResponseBytes}-byte limit`, MODEL_ERROR_CODES.TRANSPORT);
|
|
905
|
-
const idleDeadline = createStreamIdleDeadline(connection.streamIdleTimeoutMs, this.displayName, 3e4);
|
|
906
|
-
const events = parseSseBounded(boundedResponseBody(response.body, maxResponseBytes, maxResponseChunks, this.displayName, signal), idleDeadline.activity, 3e4, {
|
|
907
|
-
maxEvents: maxSseEvents,
|
|
908
|
-
maxEventChars: maxSseEventChars
|
|
909
|
-
});
|
|
910
|
-
const translated = requireTerminalFinish(this.translate(events, request), this.displayName);
|
|
911
|
-
for await (const chunk of withAbortSignal(idleDeadline.guard(translated), signal)) {
|
|
912
|
-
if (chunk.type === "usage") {
|
|
913
|
-
attemptUsage = chunk.usage;
|
|
914
|
-
const validated = validateUsageCounters(chunk.usage, true);
|
|
915
|
-
if (!validated.complete) continue;
|
|
916
|
-
yield {
|
|
917
|
-
type: "usage",
|
|
918
|
-
usage: validated.reported
|
|
919
|
-
};
|
|
920
|
-
continue;
|
|
921
|
-
}
|
|
922
|
-
if (chunk.type === "finish") {
|
|
923
|
-
attemptStatus = chunk.reason.kind === "aborted" ? "aborted" : chunk.reason.kind === "error" ? "error" : "success";
|
|
924
|
-
if (chunk.reason.kind === "error" || chunk.reason.kind === "aborted") attemptError = safeProviderFailure(chunk.reason.failure);
|
|
925
|
-
}
|
|
926
|
-
yield chunk;
|
|
927
|
-
}
|
|
928
|
-
} catch (error) {
|
|
929
|
-
if (admissionFailure !== void 0 && error === admissionFailure.value) throw error;
|
|
930
|
-
const mapped = timeout.aborted && options.signal?.aborted !== true ? new ModelError(`${this.displayName} request exceeded its ${requestTimeoutMs}ms time limit`, MODEL_ERROR_CODES.TIMEOUT, { cause: error }) : signal.aborted ? abortError(this.displayName, error) : normalizeHttpBoundaryError(error, `${this.displayName} request to ${origin} failed`);
|
|
931
|
-
attemptStatus = mapped.code === MODEL_ERROR_CODES.ABORTED ? "aborted" : "error";
|
|
932
|
-
attemptError = safeProviderFailure(mapped.failure);
|
|
933
|
-
throw mapped;
|
|
934
|
-
} finally {
|
|
935
|
-
attempt?.end({
|
|
936
|
-
status: attemptStatus,
|
|
937
|
-
dispatchState,
|
|
938
|
-
...attemptUsage === void 0 ? {} : { reported: attemptUsage },
|
|
939
|
-
...httpStatus === void 0 ? {} : { httpStatus },
|
|
940
|
-
...providerRequestId === void 0 ? {} : { providerRequestId },
|
|
941
|
-
...attemptError === void 0 ? {} : { error: attemptError }
|
|
942
|
-
});
|
|
1192
|
+
const sseLimits = resolveSseLimits(connection);
|
|
1193
|
+
const adapter = this;
|
|
1194
|
+
yield* transportStream({
|
|
1195
|
+
connection,
|
|
1196
|
+
displayName: this.displayName,
|
|
1197
|
+
provider: options.provider,
|
|
1198
|
+
model: options.model,
|
|
1199
|
+
accept: "text/event-stream",
|
|
1200
|
+
/**
|
|
1201
|
+
* Read by the transport AFTER the body is prepared, which is exactly where
|
|
1202
|
+
* the pipeline used to call it. A provider whose path computation fails
|
|
1203
|
+
* therefore still fails inside the transport's classification, and a
|
|
1204
|
+
* protocol that reports its routing decision from `endpointPath` still
|
|
1205
|
+
* reports it once, in the same place in the sequence, as before.
|
|
1206
|
+
*/
|
|
1207
|
+
get path() {
|
|
1208
|
+
return adapter.endpointPath(request);
|
|
1209
|
+
},
|
|
1210
|
+
body: (signal) => wireBodyCache.prepared ??= this.prepareWireBody(request, signal),
|
|
1211
|
+
...options.signal === void 0 ? {} : { signal: options.signal },
|
|
1212
|
+
...context === void 0 ? {} : { context },
|
|
1213
|
+
errorCode: (status, detail) => this.providerErrorCode(status, detail),
|
|
1214
|
+
observeRequest: (record) => this.observeRequest(record)
|
|
1215
|
+
}, (session) => this.decodeSse(session, request, sseLimits));
|
|
1216
|
+
}
|
|
1217
|
+
/**
|
|
1218
|
+
* The SSE half: media type, bounds, idle deadline, translation, usage honesty.
|
|
1219
|
+
*
|
|
1220
|
+
* Everything this sees has already cleared the transport's guards — 2xx, no
|
|
1221
|
+
* redirect, attempt open, teardown owned — so what remains is only the format.
|
|
1222
|
+
* Abort racing is not repeated here: {@link transportStream} already iterates
|
|
1223
|
+
* this generator under the fused signal.
|
|
1224
|
+
* @param session - the guarded response and its attempt-evidence hooks.
|
|
1225
|
+
* @param request - the request this response answers.
|
|
1226
|
+
* @param sse - event bounds resolved before any transport work began.
|
|
1227
|
+
* @returns the provider's chunks, with incomplete usage held back.
|
|
1228
|
+
*/
|
|
1229
|
+
async *decodeSse(session, request, sse) {
|
|
1230
|
+
const response = session.response;
|
|
1231
|
+
if (response.headers.get("content-type")?.split(";", 1)[0]?.trim().toLowerCase() !== session.accept) throw new ModelError(`${this.displayName} response is not text/event-stream`, HTTP_PROVIDER_ERROR_CODES.STREAM_MEDIA_TYPE_INVALID);
|
|
1232
|
+
if (response.body === null) throw new ModelError(`${this.displayName} returned no response body`, MODEL_ERROR_CODES.STREAM_CLOSED);
|
|
1233
|
+
const maxResponseBytes = session.limits.maxResponseBytes;
|
|
1234
|
+
const declaredLength = response.headers.get("content-length");
|
|
1235
|
+
if (declaredLength !== null && /^\d+$/.test(declaredLength) && Number(declaredLength) > maxResponseBytes) throw new ModelError(`${this.displayName} response exceeds the ${maxResponseBytes}-byte limit`, MODEL_ERROR_CODES.TRANSPORT);
|
|
1236
|
+
const idleDeadline = createStreamIdleDeadline(request.connection.streamIdleTimeoutMs, this.displayName, 3e4);
|
|
1237
|
+
const events = parseSseBounded(boundedResponseBody(response.body, maxResponseBytes, session.limits.maxResponseChunks, this.displayName, session.signal), idleDeadline.activity, 3e4, {
|
|
1238
|
+
maxEvents: sse.maxEvents,
|
|
1239
|
+
maxEventChars: sse.maxEventChars
|
|
1240
|
+
});
|
|
1241
|
+
const translated = requireTerminalFinish(this.translate(events, request), this.displayName);
|
|
1242
|
+
for await (const chunk of idleDeadline.guard(translated)) {
|
|
1243
|
+
if (chunk.type === "usage-progress") {
|
|
1244
|
+
session.reportUsage(chunk.usage, false);
|
|
1245
|
+
const validated = validateUsageCounters(chunk.usage, true);
|
|
1246
|
+
if (Object.keys(validated.reported).length > 0) yield {
|
|
1247
|
+
type: "usage-progress",
|
|
1248
|
+
usage: validated.reported,
|
|
1249
|
+
...session.attemptId === void 0 ? {} : { attemptId: session.attemptId }
|
|
1250
|
+
};
|
|
1251
|
+
continue;
|
|
943
1252
|
}
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
1253
|
+
if (chunk.type === "usage") {
|
|
1254
|
+
session.reportUsage(chunk.usage);
|
|
1255
|
+
const validated = validateUsageCounters(chunk.usage, true);
|
|
1256
|
+
if (!validated.complete) continue;
|
|
1257
|
+
yield {
|
|
1258
|
+
type: "usage",
|
|
1259
|
+
usage: validated.reported
|
|
1260
|
+
};
|
|
1261
|
+
continue;
|
|
1262
|
+
}
|
|
1263
|
+
if (chunk.type === "finish") {
|
|
1264
|
+
const status = chunk.reason.kind === "aborted" ? "aborted" : chunk.reason.kind === "error" ? "error" : "success";
|
|
1265
|
+
session.reportOutcome(status, chunk.reason.kind === "error" || chunk.reason.kind === "aborted" ? chunk.reason.failure : void 0);
|
|
1266
|
+
}
|
|
1267
|
+
yield chunk;
|
|
952
1268
|
}
|
|
953
1269
|
}
|
|
954
|
-
async prepareWireBody(request,
|
|
1270
|
+
async prepareWireBody(request, signal) {
|
|
955
1271
|
signal.throwIfAborted();
|
|
956
1272
|
const value = await raceWithSignal(Promise.resolve(this.buildBody(request)), signal);
|
|
957
1273
|
const encoded = JSON.stringify(value);
|
|
958
1274
|
const bytes = new TextEncoder().encode(encoded).byteLength;
|
|
959
|
-
if (bytes > maxRequestBytes) throw new ModelError(`${this.displayName} request exceeds the ${maxRequestBytes}-byte limit`, MODEL_ERROR_CODES.INVALID_REQUEST);
|
|
960
1275
|
return Object.freeze({
|
|
961
1276
|
value,
|
|
962
1277
|
encoded,
|
|
963
1278
|
bytes
|
|
964
1279
|
});
|
|
965
1280
|
}
|
|
966
|
-
/** Turn a non-2xx response into a fully populated {@link ModelError}. */
|
|
967
|
-
async httpFailure(response, url, maxBytes, signal) {
|
|
968
|
-
let raw = "";
|
|
969
|
-
try {
|
|
970
|
-
raw = await readBoundedText(response, maxBytes, signal);
|
|
971
|
-
} catch {}
|
|
972
|
-
const { message, detail } = parseErrorBody(raw);
|
|
973
|
-
const delay = retryAfterMs(response.headers.get("retry-after"));
|
|
974
|
-
const id = requestIdFrom(response.headers);
|
|
975
|
-
return new ModelError(message ?? `${this.displayName} error (HTTP ${response.status}) from ${url}`, this.providerErrorCode(response.status, detail), {
|
|
976
|
-
cause: new Error(raw.length > 0 ? raw : `HTTP ${response.status}`),
|
|
977
|
-
status: response.status,
|
|
978
|
-
...delay === void 0 ? {} : { providerRetryAfterMs: delay },
|
|
979
|
-
...id === void 0 ? {} : { requestId: id }
|
|
980
|
-
});
|
|
981
|
-
}
|
|
982
1281
|
};
|
|
983
1282
|
|
|
984
1283
|
//#endregion
|
|
@@ -1753,5 +2052,206 @@ async function* parseSse(stream, onActivity, teardownTimeoutMs = DEFAULT_SSE_TEA
|
|
|
1753
2052
|
}
|
|
1754
2053
|
|
|
1755
2054
|
//#endregion
|
|
1756
|
-
|
|
2055
|
+
//#region src/transport/embedding-connection.ts
|
|
2056
|
+
/**
|
|
2057
|
+
* Embedding's half of a connection snapshot, plus the one translation from a
|
|
2058
|
+
* declared route catalog into `Embedding_Catalog` capabilities.
|
|
2059
|
+
*
|
|
2060
|
+
* This module sits on top of {@link HttpTransportConnection} rather than inside
|
|
2061
|
+
* it: the transport and the JSON pipeline know nothing about embedding
|
|
2062
|
+
* vocabulary, and they must not, or a second pipeline could not exist without
|
|
2063
|
+
* dragging the first one's catalog along. What embedding adds is a catalog with
|
|
2064
|
+
* its own semantics — vector widths, batch ceilings, purpose handling, and a
|
|
2065
|
+
* declaration about which embedding space the vectors belong to.
|
|
2066
|
+
*
|
|
2067
|
+
* The translation rule has exactly one shape and it runs in one direction: a
|
|
2068
|
+
* field the configuration DECLARES becomes `supported`, a field it OMITS becomes
|
|
2069
|
+
* `unknown` (Requirement 10.5). Nothing here manufactures a `supported` value
|
|
2070
|
+
* from a default, because a default is the SDK's opinion, not the route's claim,
|
|
2071
|
+
* and `unknown` is never a reason to reject a request.
|
|
2072
|
+
*
|
|
2073
|
+
* Two things are deliberately NOT here:
|
|
2074
|
+
*
|
|
2075
|
+
* - Cleartext HTTP handling. `EmbeddingHttpConnection` extends the transport
|
|
2076
|
+
* snapshot, so `allowInsecureHttp` is the same field the same `endpointUrl()`
|
|
2077
|
+
* already enforces for generation — a self-hosted `http://` endpoint is
|
|
2078
|
+
* rejected unless the caller opts in explicitly, and embedding gets that for
|
|
2079
|
+
* free rather than through a second copy of the rule (Requirement 15.5).
|
|
2080
|
+
* - The generation catalog. `models`, `defaultMaxTokens` and
|
|
2081
|
+
* `defaultContextWindow` stay on `HttpConnection`; folding the two catalogs
|
|
2082
|
+
* into one shape is the conflation Requirement 10.1 rules out.
|
|
2083
|
+
*
|
|
2084
|
+
* @module ai-agent-sdk/providers/transport/embedding-connection
|
|
2085
|
+
*/
|
|
2086
|
+
/** The `unknown` capability, shared so translations need not re-allocate it. */
|
|
2087
|
+
const UNKNOWN = Object.freeze({ state: "unknown" });
|
|
2088
|
+
/** The `unsupported` capability, for a route's positive negative claim. */
|
|
2089
|
+
const UNSUPPORTED = Object.freeze({ state: "unsupported" });
|
|
2090
|
+
/** Declared value becomes `supported`; omission stays `unknown`. */
|
|
2091
|
+
function declared(value) {
|
|
2092
|
+
return value === void 0 ? UNKNOWN : Object.freeze({
|
|
2093
|
+
state: "supported",
|
|
2094
|
+
value
|
|
2095
|
+
});
|
|
2096
|
+
}
|
|
2097
|
+
/** The one place `'unsupported'` is separated from an omitted declaration. */
|
|
2098
|
+
function declaredPurpose(value) {
|
|
2099
|
+
if (value === void 0) return UNKNOWN;
|
|
2100
|
+
return value === "unsupported" ? UNSUPPORTED : Object.freeze({
|
|
2101
|
+
state: "supported",
|
|
2102
|
+
value
|
|
2103
|
+
});
|
|
2104
|
+
}
|
|
2105
|
+
/**
|
|
2106
|
+
* Translate one declared catalog entry into `Embedding_Catalog` metadata.
|
|
2107
|
+
*
|
|
2108
|
+
* `inputTypes` and `representation` come out `unknown` rather than filled in
|
|
2109
|
+
* from the v1 scope: the scope is already readable from the types
|
|
2110
|
+
* (`EmbeddingInputType` is `'text'`, `EmbeddingRepresentation` is
|
|
2111
|
+
* `'dense-float32'`), and stating `supported` on the route's behalf would be
|
|
2112
|
+
* the SDK asserting a claim the route never made.
|
|
2113
|
+
* @param provider - route that owns the entry.
|
|
2114
|
+
* @param model - the declared catalog entry.
|
|
2115
|
+
* @returns advisory embedding metadata for this entry.
|
|
2116
|
+
*/
|
|
2117
|
+
function embeddingCatalogModelInfo(provider, model) {
|
|
2118
|
+
return {
|
|
2119
|
+
provider,
|
|
2120
|
+
id: model.id,
|
|
2121
|
+
name: model.name ?? model.id,
|
|
2122
|
+
...model.description === void 0 ? {} : { description: model.description },
|
|
2123
|
+
inputTypes: UNKNOWN,
|
|
2124
|
+
representation: UNKNOWN,
|
|
2125
|
+
dimensions: declared(model.dimensions),
|
|
2126
|
+
defaultDimensions: declared(model.defaultDimensions),
|
|
2127
|
+
maxInputTokens: declared(model.maxInputTokens),
|
|
2128
|
+
maxBatchItems: declared(model.maxBatchItems),
|
|
2129
|
+
maxBatchTokens: declared(model.maxBatchTokens),
|
|
2130
|
+
maxBatchBytes: declared(model.maxBatchBytes),
|
|
2131
|
+
purposeHandling: declaredPurpose(model.purposeHandling),
|
|
2132
|
+
normalization: declared(model.normalization),
|
|
2133
|
+
compatibilityIdentity: declared(model.compatibilityIdentity)
|
|
2134
|
+
};
|
|
2135
|
+
}
|
|
2136
|
+
/**
|
|
2137
|
+
* Resolve exact embedding metadata for a model id from an advisory catalog.
|
|
2138
|
+
*
|
|
2139
|
+
* An id the catalog does not describe stays usable: the caller gets an
|
|
2140
|
+
* identity-only descriptor with every capability `unknown` and the provider
|
|
2141
|
+
* decides (Requirement 10.3). The catalog restricting the request would make
|
|
2142
|
+
* membership authoritative, which it is not.
|
|
2143
|
+
* @param provider - route being resolved.
|
|
2144
|
+
* @param modelId - requested model id.
|
|
2145
|
+
* @param models - the route's declared catalog.
|
|
2146
|
+
* @returns exact metadata for the id, or the identity-only descriptor.
|
|
2147
|
+
*/
|
|
2148
|
+
function resolvedEmbeddingCatalogModelInfo(provider, modelId, models) {
|
|
2149
|
+
const configured = models.find((entry) => entry.id === modelId);
|
|
2150
|
+
if (configured === void 0) return unknownEmbeddingModel(provider, modelId);
|
|
2151
|
+
return embeddingCatalogModelInfo(provider, configured);
|
|
2152
|
+
}
|
|
2153
|
+
|
|
2154
|
+
//#endregion
|
|
2155
|
+
//#region src/transport/json.ts
|
|
2156
|
+
/**
|
|
2157
|
+
* The JSON pipeline's entry into the shared transport chain.
|
|
2158
|
+
*
|
|
2159
|
+
* Where the streaming pipeline keeps a provider attempt open for as long as a
|
|
2160
|
+
* consumer keeps pulling, this one has a single, bounded shape: read the whole body
|
|
2161
|
+
* under the configured byte and chunk bounds, parse it, hand the parsed value to
|
|
2162
|
+
* `decode`, close the attempt. There is no framing to track and no idle deadline to
|
|
2163
|
+
* enforce, because there is nothing to wait for between events.
|
|
2164
|
+
*
|
|
2165
|
+
* Three failures are stated rather than inferred, since each one has a
|
|
2166
|
+
* plausible-looking wrong answer:
|
|
2167
|
+
*
|
|
2168
|
+
* - A response whose media type is not JSON is refused with a code of its own
|
|
2169
|
+
* ({@link HTTP_PROVIDER_ERROR_CODES.JSON_MEDIA_TYPE_INVALID}), symmetric with the
|
|
2170
|
+
* SSE check. Parsing an HTML error page as if it were the provider's answer is how
|
|
2171
|
+
* a proxy outage becomes a mysterious schema error further up.
|
|
2172
|
+
* - A body past `maxResponseBytes` is a transport failure, the same as on the SSE
|
|
2173
|
+
* path, and is refused instead of truncated: half a JSON document is not data.
|
|
2174
|
+
* - A body that is not valid JSON is a protocol failure, never something to guess
|
|
2175
|
+
* at.
|
|
2176
|
+
*
|
|
2177
|
+
* What `decode` receives is a response that already cleared every guard in
|
|
2178
|
+
* {@link withTransportSession} — 2xx, no redirect, attempt open, teardown owned —
|
|
2179
|
+
* plus a parsed body. What it owns is the schema.
|
|
2180
|
+
*
|
|
2181
|
+
* @module ai-agent-sdk/providers/transport/json
|
|
2182
|
+
*/
|
|
2183
|
+
/** Media types this pipeline accepts back from a provider. */
|
|
2184
|
+
const JSON_MEDIA_TYPES = Object.freeze(["application/json"]);
|
|
2185
|
+
/**
|
|
2186
|
+
* Send one request, read its JSON body under bound, and return `decode`'s value.
|
|
2187
|
+
*
|
|
2188
|
+
* The attempt closes as soon as the value is produced; nothing here stays open for
|
|
2189
|
+
* a consumer, because the whole response is already in memory by then.
|
|
2190
|
+
* @param input - request facts captured from one connection snapshot.
|
|
2191
|
+
* @param decode - turns a parsed body into the pipeline's own value.
|
|
2192
|
+
* @returns whatever `decode` returns.
|
|
2193
|
+
*/
|
|
2194
|
+
async function transportJson(input, decode) {
|
|
2195
|
+
const decoded = [];
|
|
2196
|
+
for await (const value of withTransportSession(input, async function* (session) {
|
|
2197
|
+
const result = await decode(session, parseJsonBody(await readBoundedBody(session, input.displayName), input.displayName));
|
|
2198
|
+
session.reportOutcome("success");
|
|
2199
|
+
yield result;
|
|
2200
|
+
})) {
|
|
2201
|
+
decoded.push({ value });
|
|
2202
|
+
break;
|
|
2203
|
+
}
|
|
2204
|
+
const first = decoded[0];
|
|
2205
|
+
if (first === void 0) throw new ModelError(`${input.displayName} produced no JSON response`, MODEL_ERROR_CODES.MALFORMED_RESPONSE);
|
|
2206
|
+
return first.value;
|
|
2207
|
+
}
|
|
2208
|
+
/** True when a `content-type` value names one of {@link JSON_MEDIA_TYPES}. */
|
|
2209
|
+
function isJsonMediaType(contentType) {
|
|
2210
|
+
const mediaType = contentType?.split(";", 1)[0]?.trim().toLowerCase();
|
|
2211
|
+
if (mediaType === void 0) return false;
|
|
2212
|
+
return JSON_MEDIA_TYPES.includes(mediaType);
|
|
2213
|
+
}
|
|
2214
|
+
/**
|
|
2215
|
+
* Read the whole body as text, refusing anything past the configured bounds.
|
|
2216
|
+
*
|
|
2217
|
+
* The declared `content-length` is checked first so an oversized response is
|
|
2218
|
+
* refused before a single byte of it is buffered; the streaming bound still applies
|
|
2219
|
+
* afterwards, because the header is a claim and not a guarantee.
|
|
2220
|
+
* @param session - the guarded response and its resolved limits.
|
|
2221
|
+
* @param displayName - provider name used in every message raised here.
|
|
2222
|
+
* @returns the exact body text.
|
|
2223
|
+
*/
|
|
2224
|
+
async function readBoundedBody(session, displayName) {
|
|
2225
|
+
const response = session.response;
|
|
2226
|
+
if (!isJsonMediaType(response.headers.get("content-type"))) throw new ModelError(`${displayName} response is not ${JSON_MEDIA_TYPES.join(" or ")}`, HTTP_PROVIDER_ERROR_CODES.JSON_MEDIA_TYPE_INVALID);
|
|
2227
|
+
if (response.body === null) throw new ModelError(`${displayName} returned no response body`, MODEL_ERROR_CODES.MALFORMED_RESPONSE);
|
|
2228
|
+
const maxResponseBytes = session.limits.maxResponseBytes;
|
|
2229
|
+
const declaredLength = response.headers.get("content-length");
|
|
2230
|
+
if (declaredLength !== null && /^\d+$/.test(declaredLength) && Number(declaredLength) > maxResponseBytes) throw new ModelError(`${displayName} response exceeds the ${maxResponseBytes}-byte limit`, MODEL_ERROR_CODES.TRANSPORT);
|
|
2231
|
+
const reader = boundedResponseBody(response.body, maxResponseBytes, session.limits.maxResponseChunks, displayName, session.signal).getReader();
|
|
2232
|
+
const decoder = new TextDecoder();
|
|
2233
|
+
let text = "";
|
|
2234
|
+
try {
|
|
2235
|
+
while (true) {
|
|
2236
|
+
const { done, value } = await raceWithSignal(reader.read(), session.signal);
|
|
2237
|
+
if (done) break;
|
|
2238
|
+
if (value === void 0) continue;
|
|
2239
|
+
text += decoder.decode(value, { stream: true });
|
|
2240
|
+
}
|
|
2241
|
+
return text + decoder.decode();
|
|
2242
|
+
} finally {
|
|
2243
|
+
reader.releaseLock();
|
|
2244
|
+
}
|
|
2245
|
+
}
|
|
2246
|
+
/** Parse a body, turning a parse failure into a protocol error rather than a guess. */
|
|
2247
|
+
function parseJsonBody(text, displayName) {
|
|
2248
|
+
try {
|
|
2249
|
+
return JSON.parse(text);
|
|
2250
|
+
} catch (error) {
|
|
2251
|
+
throw new ModelError(`${displayName} response body is not valid JSON`, MODEL_ERROR_CODES.MALFORMED_RESPONSE, { cause: error });
|
|
2252
|
+
}
|
|
2253
|
+
}
|
|
2254
|
+
|
|
2255
|
+
//#endregion
|
|
2256
|
+
export { DEFAULT_MAX_ERROR_BODY_BYTES, DEFAULT_MAX_REQUEST_BYTES, DEFAULT_MAX_RESPONSE_BYTES, DEFAULT_MAX_RESPONSE_CHUNKS, DEFAULT_REQUEST_LOGGER_TIMEOUT_MS, DEFAULT_REQUEST_TIMEOUT_MS, DEFAULT_STREAM_IDLE_TIMEOUT_MS, HTTP_PROTOCOL_API_VERSION, HTTP_PROVIDER_ERROR_CODES, HttpModelAdapter, JSON_MEDIA_TYPES, applyModelContextPolicy, captureTransportConnection, createHttpProvider, createModelContextPolicy, createRuntimeHttpProvider, defineWireProtocol, embeddingCatalogModelInfo, httpErrorCode, isJsonMediaType, observeCredentialOperation, observeModelCatalogOperation, parseErrorBody, parseSse, redactHeaders, requestIdFrom, resolveDialect, resolveTransportLimits, resolvedEmbeddingCatalogModelInfo, retryAfterMs, transportJson };
|
|
1757
2257
|
//# sourceMappingURL=index.js.map
|