@adaptic/utils 0.0.1015 → 0.0.1016
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +2823 -25
- package/dist/index.cjs.map +1 -1
- package/dist/index.mjs +2783 -25
- package/dist/index.mjs.map +1 -1
- package/dist/types/__tests__/llm/client/support/rejections.d.ts +21 -0
- package/dist/types/__tests__/llm/client/support/rejections.d.ts.map +1 -0
- package/dist/types/__tests__/llm/client/support/routes.d.ts +67 -0
- package/dist/types/__tests__/llm/client/support/routes.d.ts.map +1 -0
- package/dist/types/__tests__/llm/client/support/streams.d.ts +74 -0
- package/dist/types/__tests__/llm/client/support/streams.d.ts.map +1 -0
- package/dist/types/__tests__/llm/client/support/transports.d.ts +105 -0
- package/dist/types/__tests__/llm/client/support/transports.d.ts.map +1 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.d.ts.map +1 -1
- package/dist/types/llm/alias-client.d.ts +64 -0
- package/dist/types/llm/alias-client.d.ts.map +1 -0
- package/dist/types/llm/circuit-breaker.d.ts +124 -0
- package/dist/types/llm/circuit-breaker.d.ts.map +1 -0
- package/dist/types/llm/eval/comparators.d.ts +127 -0
- package/dist/types/llm/eval/comparators.d.ts.map +1 -0
- package/dist/types/llm/eval/coverage.d.ts +44 -0
- package/dist/types/llm/eval/coverage.d.ts.map +1 -0
- package/dist/types/llm/eval/golden-set.d.ts +47 -0
- package/dist/types/llm/eval/golden-set.d.ts.map +1 -0
- package/dist/types/llm/eval/index.d.ts +25 -0
- package/dist/types/llm/eval/index.d.ts.map +1 -0
- package/dist/types/llm/eval/json-shape.d.ts +74 -0
- package/dist/types/llm/eval/json-shape.d.ts.map +1 -0
- package/dist/types/llm/eval/judge.d.ts +131 -0
- package/dist/types/llm/eval/judge.d.ts.map +1 -0
- package/dist/types/llm/eval/metrics.d.ts +51 -0
- package/dist/types/llm/eval/metrics.d.ts.map +1 -0
- package/dist/types/llm/eval/run.d.ts +97 -0
- package/dist/types/llm/eval/run.d.ts.map +1 -0
- package/dist/types/llm/eval/types.d.ts +242 -0
- package/dist/types/llm/eval/types.d.ts.map +1 -0
- package/dist/types/llm/fallback-chain.d.ts +97 -0
- package/dist/types/llm/fallback-chain.d.ts.map +1 -0
- package/dist/types/llm/index.d.ts +30 -0
- package/dist/types/llm/index.d.ts.map +1 -0
- package/dist/types/llm/param-matrix.d.ts +65 -0
- package/dist/types/llm/param-matrix.d.ts.map +1 -0
- package/dist/types/llm/rate-guard.d.ts +119 -0
- package/dist/types/llm/rate-guard.d.ts.map +1 -0
- package/dist/types/llm/route-table.d.ts +155 -0
- package/dist/types/llm/route-table.d.ts.map +1 -0
- package/dist/types/llm/schema-retry.d.ts +80 -0
- package/dist/types/llm/schema-retry.d.ts.map +1 -0
- package/dist/types/llm/streaming.d.ts +93 -0
- package/dist/types/llm/streaming.d.ts.map +1 -0
- package/dist/types/llm/transports/direct.d.ts +92 -0
- package/dist/types/llm/transports/direct.d.ts.map +1 -0
- package/dist/types/llm/transports/gateway.d.ts +73 -0
- package/dist/types/llm/transports/gateway.d.ts.map +1 -0
- package/dist/types/llm/types.d.ts +292 -0
- package/dist/types/llm/types.d.ts.map +1 -0
- package/dist/types/schemas/alpaca-schemas.d.ts +6 -6
- package/package.json +1 -1
- package/dist/types/__tests__/alpaca-broker-error-preservation.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-broker-error-preservation.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-client-order-id.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-client-order-id.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-functions.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-functions.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-market-data-retry.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-market-data-retry.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-order-idempotency.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-order-idempotency.test.d.ts.map +0 -1
- package/dist/types/__tests__/alpaca-trading-api.test.d.ts +0 -2
- package/dist/types/__tests__/alpaca-trading-api.test.d.ts.map +0 -1
- package/dist/types/__tests__/api-endpoints.test.d.ts +0 -2
- package/dist/types/__tests__/api-endpoints.test.d.ts.map +0 -1
- package/dist/types/__tests__/asset-allocation.test.d.ts +0 -2
- package/dist/types/__tests__/asset-allocation.test.d.ts.map +0 -1
- package/dist/types/__tests__/atr.test.d.ts +0 -2
- package/dist/types/__tests__/atr.test.d.ts.map +0 -1
- package/dist/types/__tests__/auth-validator.test.d.ts +0 -2
- package/dist/types/__tests__/auth-validator.test.d.ts.map +0 -1
- package/dist/types/__tests__/broker-factory.test.d.ts +0 -2
- package/dist/types/__tests__/broker-factory.test.d.ts.map +0 -1
- package/dist/types/__tests__/broker-types.test.d.ts +0 -2
- package/dist/types/__tests__/broker-types.test.d.ts.map +0 -1
- package/dist/types/__tests__/cache.test.d.ts +0 -2
- package/dist/types/__tests__/cache.test.d.ts.map +0 -1
- package/dist/types/__tests__/errors.test.d.ts +0 -2
- package/dist/types/__tests__/errors.test.d.ts.map +0 -1
- package/dist/types/__tests__/financial-regression.test.d.ts +0 -2
- package/dist/types/__tests__/financial-regression.test.d.ts.map +0 -1
- package/dist/types/__tests__/format-tools.test.d.ts +0 -2
- package/dist/types/__tests__/format-tools.test.d.ts.map +0 -1
- package/dist/types/__tests__/http-keep-alive.test.d.ts +0 -2
- package/dist/types/__tests__/http-keep-alive.test.d.ts.map +0 -1
- package/dist/types/__tests__/http-timeout.test.d.ts +0 -2
- package/dist/types/__tests__/http-timeout.test.d.ts.map +0 -1
- package/dist/types/__tests__/index.test.d.ts +0 -2
- package/dist/types/__tests__/index.test.d.ts.map +0 -1
- package/dist/types/__tests__/legacy-auth.test.d.ts +0 -2
- package/dist/types/__tests__/legacy-auth.test.d.ts.map +0 -1
- package/dist/types/__tests__/logger.test.d.ts +0 -2
- package/dist/types/__tests__/logger.test.d.ts.map +0 -1
- package/dist/types/__tests__/logging.test.d.ts +0 -2
- package/dist/types/__tests__/logging.test.d.ts.map +0 -1
- package/dist/types/__tests__/market-time.test.d.ts +0 -2
- package/dist/types/__tests__/market-time.test.d.ts.map +0 -1
- package/dist/types/__tests__/massive.test.d.ts +0 -2
- package/dist/types/__tests__/massive.test.d.ts.map +0 -1
- package/dist/types/__tests__/metrics-calcs-direction.test.d.ts +0 -2
- package/dist/types/__tests__/metrics-calcs-direction.test.d.ts.map +0 -1
- package/dist/types/__tests__/misc-utils.test.d.ts +0 -2
- package/dist/types/__tests__/misc-utils.test.d.ts.map +0 -1
- package/dist/types/__tests__/paginator.test.d.ts +0 -2
- package/dist/types/__tests__/paginator.test.d.ts.map +0 -1
- package/dist/types/__tests__/performance-metrics-fees.test.d.ts +0 -2
- package/dist/types/__tests__/performance-metrics-fees.test.d.ts.map +0 -1
- package/dist/types/__tests__/performance-metrics.test.d.ts +0 -2
- package/dist/types/__tests__/performance-metrics.test.d.ts.map +0 -1
- package/dist/types/__tests__/price-utils-fees.test.d.ts +0 -2
- package/dist/types/__tests__/price-utils-fees.test.d.ts.map +0 -1
- package/dist/types/__tests__/price-utils.test.d.ts +0 -2
- package/dist/types/__tests__/price-utils.test.d.ts.map +0 -1
- package/dist/types/__tests__/property-based-financial.test.d.ts +0 -2
- package/dist/types/__tests__/property-based-financial.test.d.ts.map +0 -1
- package/dist/types/__tests__/protective-order-sides.test.d.ts +0 -2
- package/dist/types/__tests__/protective-order-sides.test.d.ts.map +0 -1
- package/dist/types/__tests__/rate-limiter.test.d.ts +0 -2
- package/dist/types/__tests__/rate-limiter.test.d.ts.map +0 -1
- package/dist/types/__tests__/retry-classification.test.d.ts +0 -2
- package/dist/types/__tests__/retry-classification.test.d.ts.map +0 -1
- package/dist/types/__tests__/retry.test.d.ts +0 -2
- package/dist/types/__tests__/retry.test.d.ts.map +0 -1
- package/dist/types/__tests__/risk-free-rate.test.d.ts +0 -2
- package/dist/types/__tests__/risk-free-rate.test.d.ts.map +0 -1
- package/dist/types/__tests__/risk-metrics.test.d.ts +0 -2
- package/dist/types/__tests__/risk-metrics.test.d.ts.map +0 -1
- package/dist/types/__tests__/schema-validation.test.d.ts +0 -2
- package/dist/types/__tests__/schema-validation.test.d.ts.map +0 -1
- package/dist/types/__tests__/stampede-load-timeout.test.d.ts +0 -2
- package/dist/types/__tests__/stampede-load-timeout.test.d.ts.map +0 -1
- package/dist/types/__tests__/strategy-metrics.test.d.ts +0 -2
- package/dist/types/__tests__/strategy-metrics.test.d.ts.map +0 -1
- package/dist/types/__tests__/technical-analysis-totality.test.d.ts +0 -2
- package/dist/types/__tests__/technical-analysis-totality.test.d.ts.map +0 -1
- package/dist/types/__tests__/technical-analysis.test.d.ts +0 -2
- package/dist/types/__tests__/technical-analysis.test.d.ts.map +0 -1
- package/dist/types/__tests__/time-utils.test.d.ts +0 -2
- package/dist/types/__tests__/time-utils.test.d.ts.map +0 -1
- package/dist/types/__tests__/trading-policy-schemas.test.d.ts +0 -2
- package/dist/types/__tests__/trading-policy-schemas.test.d.ts.map +0 -1
- package/dist/types/__tests__/trailing-stops-portfolio.test.d.ts +0 -2
- package/dist/types/__tests__/trailing-stops-portfolio.test.d.ts.map +0 -1
- package/dist/types/__tests__/volatility.test.d.ts +0 -2
- package/dist/types/__tests__/volatility.test.d.ts.map +0 -1
package/dist/index.cjs
CHANGED
|
@@ -3290,7 +3290,7 @@ const MIN_WAKE_DELAY_MS = 1;
|
|
|
3290
3290
|
* Number of whole tokens required to release a single queued request. The token
|
|
3291
3291
|
* bucket consumes exactly one token per admitted request.
|
|
3292
3292
|
*/
|
|
3293
|
-
const TOKENS_PER_REQUEST = 1;
|
|
3293
|
+
const TOKENS_PER_REQUEST$1 = 1;
|
|
3294
3294
|
/**
|
|
3295
3295
|
* Token bucket rate limiter implementation
|
|
3296
3296
|
*
|
|
@@ -3362,8 +3362,8 @@ class TokenBucketRateLimiter {
|
|
|
3362
3362
|
// Require a WHOLE token: refill() accrues fractionally, and admitting on
|
|
3363
3363
|
// any positive fraction would release a full request per accrual tick,
|
|
3364
3364
|
// driving the bucket negative and overrunning the configured rate.
|
|
3365
|
-
if (this.tokens >= TOKENS_PER_REQUEST) {
|
|
3366
|
-
this.tokens -= TOKENS_PER_REQUEST;
|
|
3365
|
+
if (this.tokens >= TOKENS_PER_REQUEST$1) {
|
|
3366
|
+
this.tokens -= TOKENS_PER_REQUEST$1;
|
|
3367
3367
|
logger.debug(`Rate limit token acquired for ${this.config.label}`, {
|
|
3368
3368
|
remainingTokens: this.tokens,
|
|
3369
3369
|
queueLength: this.queue.length,
|
|
@@ -3408,7 +3408,7 @@ class TokenBucketRateLimiter {
|
|
|
3408
3408
|
if (this.wakeTimer !== null || this.queue.length === 0) {
|
|
3409
3409
|
return;
|
|
3410
3410
|
}
|
|
3411
|
-
const tokensNeeded = Math.max(0, TOKENS_PER_REQUEST - this.tokens);
|
|
3411
|
+
const tokensNeeded = Math.max(0, TOKENS_PER_REQUEST$1 - this.tokens);
|
|
3412
3412
|
const deficitMs = Math.max(MIN_WAKE_DELAY_MS, Math.ceil((tokensNeeded / this.config.refillRate) * MS_PER_SECOND$1));
|
|
3413
3413
|
const timer = setTimeout(() => {
|
|
3414
3414
|
this.wakeTimer = null;
|
|
@@ -3460,8 +3460,8 @@ class TokenBucketRateLimiter {
|
|
|
3460
3460
|
const logger = getLogger();
|
|
3461
3461
|
try {
|
|
3462
3462
|
// Whole-token admission — see the matching guard in acquire().
|
|
3463
|
-
while (this.queue.length > 0 && this.tokens >= TOKENS_PER_REQUEST) {
|
|
3464
|
-
this.tokens -= TOKENS_PER_REQUEST;
|
|
3463
|
+
while (this.queue.length > 0 && this.tokens >= TOKENS_PER_REQUEST$1) {
|
|
3464
|
+
this.tokens -= TOKENS_PER_REQUEST$1;
|
|
3465
3465
|
const next = this.queue.shift();
|
|
3466
3466
|
if (next) {
|
|
3467
3467
|
clearTimeout(next.timeoutHandle);
|
|
@@ -3543,7 +3543,7 @@ class TokenBucketRateLimiter {
|
|
|
3543
3543
|
* await rateLimiters.alphaVantage.acquire();
|
|
3544
3544
|
* ```
|
|
3545
3545
|
*/
|
|
3546
|
-
const rateLimiters = {
|
|
3546
|
+
const rateLimiters$1 = {
|
|
3547
3547
|
/**
|
|
3548
3548
|
* Alpaca API rate limiter
|
|
3549
3549
|
*
|
|
@@ -4149,7 +4149,7 @@ class AlpacaMarketDataAPI extends require$$0$1.EventEmitter {
|
|
|
4149
4149
|
// bucket so concurrent callers (bar fetches, quotes, options, snapshots)
|
|
4150
4150
|
// can't overrun Alpaca's server-side rate limit. Prior to this, parallel
|
|
4151
4151
|
// historical-bar fan-out produced ~125 server-side 429s per minute.
|
|
4152
|
-
await rateLimiters.alpaca.acquire();
|
|
4152
|
+
await rateLimiters$1.alpaca.acquire();
|
|
4153
4153
|
// Retry ONLY transient connection faults, and only on GET (every
|
|
4154
4154
|
// market-data read here is idempotent). A non-2xx response is a real
|
|
4155
4155
|
// answer from Alpaca and is never retried — that path still throws on
|
|
@@ -4195,7 +4195,7 @@ class AlpacaMarketDataAPI extends require$$0$1.EventEmitter {
|
|
|
4195
4195
|
await new Promise((resolve) => setTimeout(resolve, delayMs));
|
|
4196
4196
|
// Re-acquire the rate-limit token so a retry storm cannot overrun
|
|
4197
4197
|
// Alpaca's server-side limit.
|
|
4198
|
-
await rateLimiters.alpaca.acquire();
|
|
4198
|
+
await rateLimiters$1.alpaca.acquire();
|
|
4199
4199
|
}
|
|
4200
4200
|
}
|
|
4201
4201
|
if (!response) {
|
|
@@ -9902,7 +9902,7 @@ const fetchTickerInfo = async (symbol, options) => {
|
|
|
9902
9902
|
apiKey,
|
|
9903
9903
|
});
|
|
9904
9904
|
return massiveLimit(async () => {
|
|
9905
|
-
await rateLimiters.massive.acquire();
|
|
9905
|
+
await rateLimiters$1.massive.acquire();
|
|
9906
9906
|
try {
|
|
9907
9907
|
const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
|
|
9908
9908
|
const data = await response.json();
|
|
@@ -10015,7 +10015,7 @@ const fetchLastTradeImpl = async (symbol, options) => {
|
|
|
10015
10015
|
order: "desc",
|
|
10016
10016
|
});
|
|
10017
10017
|
return massiveLimit(async () => {
|
|
10018
|
-
await rateLimiters.massive.acquire();
|
|
10018
|
+
await rateLimiters$1.massive.acquire();
|
|
10019
10019
|
try {
|
|
10020
10020
|
const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
|
|
10021
10021
|
const data = (await response.json());
|
|
@@ -10086,7 +10086,7 @@ const fetchLastQuote = async (symbol, options) => {
|
|
|
10086
10086
|
order: "desc",
|
|
10087
10087
|
});
|
|
10088
10088
|
return massiveLimit(async () => {
|
|
10089
|
-
await rateLimiters.massive.acquire();
|
|
10089
|
+
await rateLimiters$1.massive.acquire();
|
|
10090
10090
|
try {
|
|
10091
10091
|
const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
|
|
10092
10092
|
const data = (await response.json());
|
|
@@ -10175,7 +10175,7 @@ const fetchPrices = async (params, options) => {
|
|
|
10175
10175
|
let aggregatedStatus = "OK";
|
|
10176
10176
|
while (nextUrl) {
|
|
10177
10177
|
//getLogger().info(`Debug: Fetching ${nextUrl}`);
|
|
10178
|
-
await rateLimiters.massive.acquire();
|
|
10178
|
+
await rateLimiters$1.massive.acquire();
|
|
10179
10179
|
const response = await fetchWithRetry(nextUrl, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
|
|
10180
10180
|
const data = await response.json();
|
|
10181
10181
|
if (!MASSIVE_VALID_STATUSES.has(data.status)) {
|
|
@@ -10362,7 +10362,7 @@ const fetchGroupedDaily = async (date, options) => {
|
|
|
10362
10362
|
include_otc: options?.includeOTC ? "true" : "false",
|
|
10363
10363
|
});
|
|
10364
10364
|
return massiveLimit(async () => {
|
|
10365
|
-
await rateLimiters.massive.acquire();
|
|
10365
|
+
await rateLimiters$1.massive.acquire();
|
|
10366
10366
|
try {
|
|
10367
10367
|
const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
|
|
10368
10368
|
const data = await response.json();
|
|
@@ -10456,7 +10456,7 @@ symbol, date = new Date(), options) => {
|
|
|
10456
10456
|
adjusted: (options?.adjusted ?? true).toString(),
|
|
10457
10457
|
});
|
|
10458
10458
|
return massiveLimit(async () => {
|
|
10459
|
-
await rateLimiters.massive.acquire();
|
|
10459
|
+
await rateLimiters$1.massive.acquire();
|
|
10460
10460
|
try {
|
|
10461
10461
|
const response = await fetchWithRetry(`${baseUrl}?${params.toString()}`, { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 1000);
|
|
10462
10462
|
const data = await response.json();
|
|
@@ -10530,7 +10530,7 @@ const fetchTrades = async (symbol, options) => {
|
|
|
10530
10530
|
if (options?.sort)
|
|
10531
10531
|
params.append("sort", options.sort);
|
|
10532
10532
|
return massiveLimit(async () => {
|
|
10533
|
-
await rateLimiters.massive.acquire();
|
|
10533
|
+
await rateLimiters$1.massive.acquire();
|
|
10534
10534
|
const url = `${baseUrl}?${params.toString()}`;
|
|
10535
10535
|
try {
|
|
10536
10536
|
logIfDebug(`Fetching trades for ${symbol} from ${url}`);
|
|
@@ -10613,7 +10613,7 @@ const fetchIndicesAggregates = async (params, options) => {
|
|
|
10613
10613
|
}
|
|
10614
10614
|
url.search = queryParams.toString();
|
|
10615
10615
|
return massiveIndicesLimit(async () => {
|
|
10616
|
-
await rateLimiters.massive.acquire();
|
|
10616
|
+
await rateLimiters$1.massive.acquire();
|
|
10617
10617
|
try {
|
|
10618
10618
|
const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
|
|
10619
10619
|
const data = await response.json();
|
|
@@ -10643,7 +10643,7 @@ const fetchIndicesPreviousClose = async (indicesTicker, options) => {
|
|
|
10643
10643
|
queryParams.append("apiKey", apiKey);
|
|
10644
10644
|
url.search = queryParams.toString();
|
|
10645
10645
|
return massiveIndicesLimit(async () => {
|
|
10646
|
-
await rateLimiters.massive.acquire();
|
|
10646
|
+
await rateLimiters$1.massive.acquire();
|
|
10647
10647
|
try {
|
|
10648
10648
|
const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
|
|
10649
10649
|
const data = await response.json();
|
|
@@ -10674,7 +10674,7 @@ const fetchIndicesDailyOpenClose = async (indicesTicker, date, options) => {
|
|
|
10674
10674
|
queryParams.append("apiKey", apiKey);
|
|
10675
10675
|
url.search = queryParams.toString();
|
|
10676
10676
|
return massiveIndicesLimit(async () => {
|
|
10677
|
-
await rateLimiters.massive.acquire();
|
|
10677
|
+
await rateLimiters$1.massive.acquire();
|
|
10678
10678
|
try {
|
|
10679
10679
|
const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
|
|
10680
10680
|
const data = await response.json();
|
|
@@ -10716,7 +10716,7 @@ const fetchIndicesSnapshot = async (params, options) => {
|
|
|
10716
10716
|
}
|
|
10717
10717
|
url.search = queryParams.toString();
|
|
10718
10718
|
return massiveIndicesLimit(async () => {
|
|
10719
|
-
await rateLimiters.massive.acquire();
|
|
10719
|
+
await rateLimiters$1.massive.acquire();
|
|
10720
10720
|
try {
|
|
10721
10721
|
const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
|
|
10722
10722
|
const data = await response.json();
|
|
@@ -10765,7 +10765,7 @@ const fetchUniversalSnapshot = async (tickers, options) => {
|
|
|
10765
10765
|
}
|
|
10766
10766
|
url.search = queryParams.toString();
|
|
10767
10767
|
return massiveIndicesLimit(async () => {
|
|
10768
|
-
await rateLimiters.massive.acquire();
|
|
10768
|
+
await rateLimiters$1.massive.acquire();
|
|
10769
10769
|
try {
|
|
10770
10770
|
const response = await fetchWithRetry(url.toString(), { signal: createTimeoutSignal(DEFAULT_TIMEOUTS.MASSIVE_API) }, 3, 300);
|
|
10771
10771
|
const data = await response.json();
|
|
@@ -52147,7 +52147,7 @@ class AlpacaClient {
|
|
|
52147
52147
|
* @returns Result of the operation
|
|
52148
52148
|
*/
|
|
52149
52149
|
async executeWithRateLimit(operation, label) {
|
|
52150
|
-
await rateLimiters.alpaca.acquire();
|
|
52150
|
+
await rateLimiters$1.alpaca.acquire();
|
|
52151
52151
|
return withRetry(operation, {
|
|
52152
52152
|
maxRetries: 2,
|
|
52153
52153
|
baseDelayMs: 1000,
|
|
@@ -52206,7 +52206,7 @@ class AlpacaClient {
|
|
|
52206
52206
|
* @returns Response data
|
|
52207
52207
|
*/
|
|
52208
52208
|
async makeRequest(endpoint, method = "GET", body) {
|
|
52209
|
-
await rateLimiters.alpaca.acquire();
|
|
52209
|
+
await rateLimiters$1.alpaca.acquire();
|
|
52210
52210
|
const url = `${this.apiBaseUrl}${endpoint}`;
|
|
52211
52211
|
const options = {
|
|
52212
52212
|
method,
|
|
@@ -62911,6 +62911,2764 @@ const alpaca = {
|
|
|
62911
62911
|
streams: streams$1,
|
|
62912
62912
|
};
|
|
62913
62913
|
|
|
62914
|
+
/**
|
|
62915
|
+
* Per-route circuit breaker for the alias client.
|
|
62916
|
+
*
|
|
62917
|
+
* A provider that has just failed five calls will almost certainly fail the
|
|
62918
|
+
* sixth, and every attempt spends latency budget the next leg of the chain
|
|
62919
|
+
* needs. The breaker converts that repeated discovery into a decision made
|
|
62920
|
+
* once: an unhealthy leg is skipped outright until it has had time to recover,
|
|
62921
|
+
* so a chain reaches its working leg quickly instead of paying the full
|
|
62922
|
+
* timeout of each dead one first.
|
|
62923
|
+
*
|
|
62924
|
+
* Breakers are keyed by chain POSITION, not by provider. Two consequences
|
|
62925
|
+
* follow, and both are intended. Reverting an alias to a different model at the
|
|
62926
|
+
* same position inherits that position's health rather than starting blind. And
|
|
62927
|
+
* an isolated route never shares a breaker with the shared one, so isolated
|
|
62928
|
+
* traffic can neither trip nor be tripped by traffic on the other side of the
|
|
62929
|
+
* PD-9 boundary.
|
|
62930
|
+
*
|
|
62931
|
+
* The clock is injected. Breaker behaviour is entirely about elapsed time, and
|
|
62932
|
+
* a test that must sleep to observe a cooldown is a test nobody runs.
|
|
62933
|
+
*
|
|
62934
|
+
* @module llm/circuit-breaker
|
|
62935
|
+
*/
|
|
62936
|
+
/**
|
|
62937
|
+
* Tracks route health and decides whether a leg may be attempted.
|
|
62938
|
+
*/
|
|
62939
|
+
class CircuitBreakerRegistry {
|
|
62940
|
+
records = new Map();
|
|
62941
|
+
config;
|
|
62942
|
+
now;
|
|
62943
|
+
/**
|
|
62944
|
+
* @param config Failure threshold, cooldown and half-open probe budget.
|
|
62945
|
+
* @param now Clock, injected so cooldowns are testable without waiting.
|
|
62946
|
+
*/
|
|
62947
|
+
constructor(config, now = Date.now) {
|
|
62948
|
+
this.config = config;
|
|
62949
|
+
this.now = now;
|
|
62950
|
+
}
|
|
62951
|
+
/**
|
|
62952
|
+
* Current state of a route's breaker.
|
|
62953
|
+
*
|
|
62954
|
+
* The transition from open to half-open is computed from elapsed time at read
|
|
62955
|
+
* time rather than scheduled with a timer. A timer would keep the process
|
|
62956
|
+
* awake for every route that ever failed, and would drift whenever the
|
|
62957
|
+
* process was busy — which is exactly when a breaker matters most.
|
|
62958
|
+
*
|
|
62959
|
+
* @param routeKey The route's stable key.
|
|
62960
|
+
* @returns Its state.
|
|
62961
|
+
*/
|
|
62962
|
+
stateOf(routeKey) {
|
|
62963
|
+
const record = this.records.get(routeKey);
|
|
62964
|
+
if (record === undefined || record.openedAtMs === null) {
|
|
62965
|
+
return "closed";
|
|
62966
|
+
}
|
|
62967
|
+
const elapsed = this.now() - record.openedAtMs;
|
|
62968
|
+
return elapsed >= this.config.cooldown_ms ? "half-open" : "open";
|
|
62969
|
+
}
|
|
62970
|
+
/**
|
|
62971
|
+
* Whether a route may be attempted now.
|
|
62972
|
+
*
|
|
62973
|
+
* A half-open route admits a bounded number of probes at once. Letting the
|
|
62974
|
+
* whole queue through the moment a cooldown expires would re-hammer a
|
|
62975
|
+
* provider that is still recovering, which is how a breaker turns into a
|
|
62976
|
+
* synchronised retry storm.
|
|
62977
|
+
*
|
|
62978
|
+
* @param routeKey The route's stable key.
|
|
62979
|
+
* @returns Whether an attempt is permitted.
|
|
62980
|
+
*/
|
|
62981
|
+
allows(routeKey) {
|
|
62982
|
+
const state = this.stateOf(routeKey);
|
|
62983
|
+
if (state === "closed") {
|
|
62984
|
+
return true;
|
|
62985
|
+
}
|
|
62986
|
+
if (state === "open") {
|
|
62987
|
+
return false;
|
|
62988
|
+
}
|
|
62989
|
+
const record = this.recordFor(routeKey);
|
|
62990
|
+
return record.probesInFlight < this.config.half_open_probes;
|
|
62991
|
+
}
|
|
62992
|
+
/**
|
|
62993
|
+
* Register that an attempt is starting, so half-open probes stay bounded.
|
|
62994
|
+
*
|
|
62995
|
+
* @param routeKey The route's stable key.
|
|
62996
|
+
* @returns void
|
|
62997
|
+
*/
|
|
62998
|
+
onAttemptStart(routeKey) {
|
|
62999
|
+
if (this.stateOf(routeKey) === "half-open") {
|
|
63000
|
+
this.recordFor(routeKey).probesInFlight += 1;
|
|
63001
|
+
}
|
|
63002
|
+
}
|
|
63003
|
+
/**
|
|
63004
|
+
* Record a success, closing the breaker.
|
|
63005
|
+
*
|
|
63006
|
+
* A single success closes it fully rather than decrementing the failure
|
|
63007
|
+
* count. The breaker's question is "is this route working now", and one
|
|
63008
|
+
* working call answers it; requiring several would keep a recovered provider
|
|
63009
|
+
* excluded while the chain paid for slower legs.
|
|
63010
|
+
*
|
|
63011
|
+
* @param routeKey The route's stable key.
|
|
63012
|
+
* @returns void
|
|
63013
|
+
*/
|
|
63014
|
+
onSuccess(routeKey) {
|
|
63015
|
+
this.records.set(routeKey, {
|
|
63016
|
+
consecutiveFailures: 0,
|
|
63017
|
+
openedAtMs: null,
|
|
63018
|
+
probesInFlight: 0,
|
|
63019
|
+
});
|
|
63020
|
+
}
|
|
63021
|
+
/**
|
|
63022
|
+
* Record a failure, opening the breaker once the threshold is reached.
|
|
63023
|
+
*
|
|
63024
|
+
* A failure while half-open re-opens immediately without waiting to
|
|
63025
|
+
* re-accumulate the threshold: the probe was the test, and it failed.
|
|
63026
|
+
*
|
|
63027
|
+
* @param routeKey The route's stable key.
|
|
63028
|
+
* @returns void
|
|
63029
|
+
*/
|
|
63030
|
+
onFailure(routeKey) {
|
|
63031
|
+
const wasHalfOpen = this.stateOf(routeKey) === "half-open";
|
|
63032
|
+
const record = this.recordFor(routeKey);
|
|
63033
|
+
record.probesInFlight = 0;
|
|
63034
|
+
record.consecutiveFailures += 1;
|
|
63035
|
+
if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
|
|
63036
|
+
record.openedAtMs = this.now();
|
|
63037
|
+
}
|
|
63038
|
+
}
|
|
63039
|
+
/**
|
|
63040
|
+
* Inspect a route's breaker.
|
|
63041
|
+
*
|
|
63042
|
+
* @param routeKey The route's stable key.
|
|
63043
|
+
* @returns A snapshot.
|
|
63044
|
+
*/
|
|
63045
|
+
snapshot(routeKey) {
|
|
63046
|
+
const record = this.records.get(routeKey) ?? {
|
|
63047
|
+
consecutiveFailures: 0,
|
|
63048
|
+
openedAtMs: null,
|
|
63049
|
+
probesInFlight: 0,
|
|
63050
|
+
};
|
|
63051
|
+
return {
|
|
63052
|
+
routeKey,
|
|
63053
|
+
state: this.stateOf(routeKey),
|
|
63054
|
+
consecutiveFailures: record.consecutiveFailures,
|
|
63055
|
+
openedAtMs: record.openedAtMs,
|
|
63056
|
+
probesInFlight: record.probesInFlight,
|
|
63057
|
+
};
|
|
63058
|
+
}
|
|
63059
|
+
/**
|
|
63060
|
+
* Every route the registry has observed.
|
|
63061
|
+
*
|
|
63062
|
+
* @returns Snapshots, sorted by route key for stable output.
|
|
63063
|
+
*/
|
|
63064
|
+
snapshotAll() {
|
|
63065
|
+
return [...this.records.keys()]
|
|
63066
|
+
.sort()
|
|
63067
|
+
.map((routeKey) => this.snapshot(routeKey));
|
|
63068
|
+
}
|
|
63069
|
+
/**
|
|
63070
|
+
* Discard all breaker state.
|
|
63071
|
+
*
|
|
63072
|
+
* @returns void
|
|
63073
|
+
*/
|
|
63074
|
+
reset() {
|
|
63075
|
+
this.records.clear();
|
|
63076
|
+
}
|
|
63077
|
+
/**
|
|
63078
|
+
* @param routeKey The route's stable key.
|
|
63079
|
+
* @returns The mutable record, created on first use.
|
|
63080
|
+
*/
|
|
63081
|
+
recordFor(routeKey) {
|
|
63082
|
+
let record = this.records.get(routeKey);
|
|
63083
|
+
if (record === undefined) {
|
|
63084
|
+
record = { consecutiveFailures: 0, openedAtMs: null, probesInFlight: 0 };
|
|
63085
|
+
this.records.set(routeKey, record);
|
|
63086
|
+
}
|
|
63087
|
+
return record;
|
|
63088
|
+
}
|
|
63089
|
+
}
|
|
63090
|
+
|
|
63091
|
+
/**
|
|
63092
|
+
* Parameter normalisation across providers.
|
|
63093
|
+
*
|
|
63094
|
+
* A caller passes one option bag and the chain may serve it from any of three
|
|
63095
|
+
* providers, so the same request has to be expressible to all of them. Vendors
|
|
63096
|
+
* disagree about more than spelling: some reject a sampling parameter they do
|
|
63097
|
+
* not support with a hard 400 rather than ignoring it, some name the output
|
|
63098
|
+
* cap differently, and reasoning models take an effort knob that non-reasoning
|
|
63099
|
+
* models refuse. Left unnormalised, a fallback would fail on the leg it fell
|
|
63100
|
+
* back to — turning the mechanism that exists to survive an outage into a
|
|
63101
|
+
* second way to fail.
|
|
63102
|
+
*
|
|
63103
|
+
* The rule throughout is that an unsupported parameter is OMITTED, never sent
|
|
63104
|
+
* with a default. Sending a default asserts a value the caller did not choose;
|
|
63105
|
+
* omitting it lets the provider apply its own, which is what "unsupported"
|
|
63106
|
+
* actually means.
|
|
63107
|
+
*
|
|
63108
|
+
* @module llm/param-matrix
|
|
63109
|
+
*/
|
|
63110
|
+
/**
|
|
63111
|
+
* Providers that name the output cap `max_tokens` rather than
|
|
63112
|
+
* `max_completion_tokens`.
|
|
63113
|
+
*
|
|
63114
|
+
* The split is a wire-format fact about each API, so it is recorded as data
|
|
63115
|
+
* next to the translation that uses it rather than inferred from a version
|
|
63116
|
+
* number that will move.
|
|
63117
|
+
*/
|
|
63118
|
+
const MAX_TOKENS_PARAM_BY_API_STYLE = {
|
|
63119
|
+
anthropic: "max_tokens",
|
|
63120
|
+
"openai-compatible": "max_completion_tokens",
|
|
63121
|
+
};
|
|
63122
|
+
/** The response-format key an OpenAI-compatible provider expects. */
|
|
63123
|
+
const RESPONSE_FORMAT_KEY = "response_format";
|
|
63124
|
+
/**
|
|
63125
|
+
* Normalise a caller's options into the exact parameter set one leg accepts.
|
|
63126
|
+
*
|
|
63127
|
+
* @param options The caller's options.
|
|
63128
|
+
* @param route The leg the request is being prepared for.
|
|
63129
|
+
* @param responseFormat The response shape the caller asked for.
|
|
63130
|
+
* @returns Parameters ready to send verbatim.
|
|
63131
|
+
*/
|
|
63132
|
+
function normaliseParams(options, route, responseFormat) {
|
|
63133
|
+
const params = {};
|
|
63134
|
+
const capabilities = route.params;
|
|
63135
|
+
// Temperature. A model that accepts only its default value returns a hard
|
|
63136
|
+
// error for the parameter's mere presence, so support is checked before the
|
|
63137
|
+
// caller's preference is consulted at all.
|
|
63138
|
+
const temperature = options.temperature ?? capabilities.temperature ?? undefined;
|
|
63139
|
+
if (capabilities.supports_temperature !== false && typeof temperature === "number") {
|
|
63140
|
+
params.temperature = temperature;
|
|
63141
|
+
}
|
|
63142
|
+
// Output cap. The caller's value wins where present; the route's declared cap
|
|
63143
|
+
// is the ceiling, because exceeding it is a provider error rather than a
|
|
63144
|
+
// larger answer.
|
|
63145
|
+
const routeCap = capabilities.max_output_tokens ?? undefined;
|
|
63146
|
+
const requested = options.maxOutputTokens ?? routeCap;
|
|
63147
|
+
if (typeof requested === "number") {
|
|
63148
|
+
const capped = typeof routeCap === "number" ? Math.min(requested, routeCap) : requested;
|
|
63149
|
+
const key = MAX_TOKENS_PARAM_BY_API_STYLE[route.provider.api_style] ?? "max_tokens";
|
|
63150
|
+
params[key] = capped;
|
|
63151
|
+
}
|
|
63152
|
+
// Reasoning effort is meaningful only where the route declares it, and is
|
|
63153
|
+
// dropped elsewhere rather than translated into a temperature.
|
|
63154
|
+
const effort = options.reasoningEffort ?? capabilities.reasoning_effort ?? undefined;
|
|
63155
|
+
if (typeof effort === "string" && capabilities.reasoning_effort !== undefined) {
|
|
63156
|
+
params.reasoning_effort = effort;
|
|
63157
|
+
}
|
|
63158
|
+
const format = normaliseResponseFormat(responseFormat, route);
|
|
63159
|
+
if (format !== undefined) {
|
|
63160
|
+
params[RESPONSE_FORMAT_KEY] = format;
|
|
63161
|
+
}
|
|
63162
|
+
if (options.tools !== undefined && options.tools.length > 0) {
|
|
63163
|
+
if (capabilities.supports_tools === false) {
|
|
63164
|
+
throw new UnsupportedCapabilityError(route, "tools");
|
|
63165
|
+
}
|
|
63166
|
+
params.tools = [...options.tools];
|
|
63167
|
+
// Parallel tool calls are off by default: a decision path that fans out
|
|
63168
|
+
// tool calls concurrently reorders its own effects, and ordering is part
|
|
63169
|
+
// of the meaning of a sequence of trading actions.
|
|
63170
|
+
params.parallel_tool_calls = false;
|
|
63171
|
+
}
|
|
63172
|
+
if (options.metadata !== undefined) {
|
|
63173
|
+
params.metadata = { ...options.metadata };
|
|
63174
|
+
}
|
|
63175
|
+
return params;
|
|
63176
|
+
}
|
|
63177
|
+
/**
|
|
63178
|
+
* Translate the requested response shape into the parameter a route accepts.
|
|
63179
|
+
*
|
|
63180
|
+
* A strict JSON schema is a real capability, not a formatting preference: a
|
|
63181
|
+
* route that cannot enforce one would return prose where the caller's parser
|
|
63182
|
+
* expects an object. Rather than silently degrading to free JSON — which fails
|
|
63183
|
+
* later, further from the cause — a route without the capability refuses the
|
|
63184
|
+
* request so the chain advances to one that has it.
|
|
63185
|
+
*
|
|
63186
|
+
* @param responseFormat What the caller asked for.
|
|
63187
|
+
* @param route The leg being prepared.
|
|
63188
|
+
* @returns The provider-facing value, or undefined for plain text.
|
|
63189
|
+
*/
|
|
63190
|
+
function normaliseResponseFormat(responseFormat, route) {
|
|
63191
|
+
if (responseFormat === "text") {
|
|
63192
|
+
return undefined;
|
|
63193
|
+
}
|
|
63194
|
+
if (responseFormat === "json") {
|
|
63195
|
+
return { type: "json_object" };
|
|
63196
|
+
}
|
|
63197
|
+
if (route.params.supports_json_schema === false) {
|
|
63198
|
+
throw new UnsupportedCapabilityError(route, "json_schema");
|
|
63199
|
+
}
|
|
63200
|
+
return {
|
|
63201
|
+
type: "json_schema",
|
|
63202
|
+
json_schema: {
|
|
63203
|
+
name: "structured_response",
|
|
63204
|
+
strict: true,
|
|
63205
|
+
schema: responseFormat.schema,
|
|
63206
|
+
},
|
|
63207
|
+
};
|
|
63208
|
+
}
|
|
63209
|
+
/**
|
|
63210
|
+
* Thrown when a leg cannot honour a capability the caller requires.
|
|
63211
|
+
*
|
|
63212
|
+
* A distinct type rather than a generic error, because the chain treats it
|
|
63213
|
+
* differently from a provider outage: the leg is not broken, it is simply the
|
|
63214
|
+
* wrong leg for this request, and no retry against it will help.
|
|
63215
|
+
*/
|
|
63216
|
+
class UnsupportedCapabilityError extends Error {
|
|
63217
|
+
/** The leg that cannot serve the request. */
|
|
63218
|
+
routeKey;
|
|
63219
|
+
/** The capability it lacks. */
|
|
63220
|
+
capability;
|
|
63221
|
+
/**
|
|
63222
|
+
* @param route The leg.
|
|
63223
|
+
* @param capability The missing capability.
|
|
63224
|
+
*/
|
|
63225
|
+
constructor(route, capability) {
|
|
63226
|
+
super(`route ${route.routeKey} (${route.providerName}/${route.modelId}) does not support ${capability}; ` +
|
|
63227
|
+
"the chain advances rather than degrading the request, so the caller's contract is never silently weakened");
|
|
63228
|
+
this.name = "UnsupportedCapabilityError";
|
|
63229
|
+
this.routeKey = route.routeKey;
|
|
63230
|
+
this.capability = capability;
|
|
63231
|
+
}
|
|
63232
|
+
}
|
|
63233
|
+
/**
|
|
63234
|
+
* Whether a leg can serve a request needing the given capabilities at all.
|
|
63235
|
+
*
|
|
63236
|
+
* Used to skip a leg before spending a network round trip on it. Checking
|
|
63237
|
+
* up front rather than reacting to the provider's rejection keeps a
|
|
63238
|
+
* capability mismatch from consuming the caller's latency budget.
|
|
63239
|
+
*
|
|
63240
|
+
* @param route The leg.
|
|
63241
|
+
* @param needs Capabilities the request requires.
|
|
63242
|
+
* @returns Whether the leg is a candidate.
|
|
63243
|
+
*/
|
|
63244
|
+
function routeSupports(route, needs) {
|
|
63245
|
+
if (needs.tools === true && route.params.supports_tools === false) {
|
|
63246
|
+
return false;
|
|
63247
|
+
}
|
|
63248
|
+
if (needs.jsonSchema === true && route.params.supports_json_schema === false) {
|
|
63249
|
+
return false;
|
|
63250
|
+
}
|
|
63251
|
+
if (needs.vision === true && route.params.supports_vision !== true) {
|
|
63252
|
+
return false;
|
|
63253
|
+
}
|
|
63254
|
+
if (needs.cacheControl === true && route.params.supports_cache_control !== true) {
|
|
63255
|
+
return false;
|
|
63256
|
+
}
|
|
63257
|
+
return true;
|
|
63258
|
+
}
|
|
63259
|
+
|
|
63260
|
+
var defaults$1 = {
|
|
63261
|
+
_readme: "Applied to any provider whose published ceiling is unknown. Chosen to be comfortably below the slowest plausible published limit: being slower than necessary costs latency, while being faster than permitted costs 429s that the fallback chain will read as provider ill-health and use to open a circuit breaker. The asymmetry is what makes the conservative side the correct default.",
|
|
63262
|
+
basis: "conservative-default",
|
|
63263
|
+
requests_per_minute: 60,
|
|
63264
|
+
max_concurrent: 4,
|
|
63265
|
+
acquire_timeout_ms: 15000
|
|
63266
|
+
};
|
|
63267
|
+
var providers$1 = {
|
|
63268
|
+
anthropic: {
|
|
63269
|
+
basis: "conservative-default",
|
|
63270
|
+
requests_per_minute: 50,
|
|
63271
|
+
max_concurrent: 4,
|
|
63272
|
+
acquire_timeout_ms: 15000,
|
|
63273
|
+
source: null,
|
|
63274
|
+
note: "Anthropic publishes tier-dependent limits; the account's tier is not recorded here. Transcribe the real ceiling from the console at W3-07 and set basis to published."
|
|
63275
|
+
},
|
|
63276
|
+
openai: {
|
|
63277
|
+
basis: "conservative-default",
|
|
63278
|
+
requests_per_minute: 60,
|
|
63279
|
+
max_concurrent: 4,
|
|
63280
|
+
acquire_timeout_ms: 15000,
|
|
63281
|
+
source: null,
|
|
63282
|
+
note: "Tier-dependent. No alias routes here today; the entry exists so a revert to an OpenAI incumbent inherits a bounded client rather than an unbounded one."
|
|
63283
|
+
},
|
|
63284
|
+
deepseek: {
|
|
63285
|
+
basis: "conservative-default",
|
|
63286
|
+
requests_per_minute: 60,
|
|
63287
|
+
max_concurrent: 4,
|
|
63288
|
+
acquire_timeout_ms: 30000,
|
|
63289
|
+
source: null,
|
|
63290
|
+
note: "Serves llm.extract, a batch-class alias, so a longer acquire timeout is appropriate: a batch caller can afford to queue where a hot-path caller cannot."
|
|
63291
|
+
},
|
|
63292
|
+
deepinfra: {
|
|
63293
|
+
basis: "conservative-default",
|
|
63294
|
+
requests_per_minute: 60,
|
|
63295
|
+
max_concurrent: 4,
|
|
63296
|
+
acquire_timeout_ms: 15000,
|
|
63297
|
+
source: null,
|
|
63298
|
+
note: "Awaiting W3-02. Transcribe the published ceiling then."
|
|
63299
|
+
},
|
|
63300
|
+
fireworks: {
|
|
63301
|
+
basis: "conservative-default",
|
|
63302
|
+
requests_per_minute: 60,
|
|
63303
|
+
max_concurrent: 4,
|
|
63304
|
+
acquire_timeout_ms: 15000,
|
|
63305
|
+
source: null,
|
|
63306
|
+
note: "The backlog calls out recording Fireworks' published RPM ceiling specifically (W3-03 -> W4-04). Do that at onboarding."
|
|
63307
|
+
},
|
|
63308
|
+
zai: {
|
|
63309
|
+
basis: "conservative-default",
|
|
63310
|
+
requests_per_minute: 60,
|
|
63311
|
+
max_concurrent: 4,
|
|
63312
|
+
acquire_timeout_ms: 15000,
|
|
63313
|
+
source: null,
|
|
63314
|
+
note: "Awaiting W3-04."
|
|
63315
|
+
},
|
|
63316
|
+
groq: {
|
|
63317
|
+
basis: "conservative-default",
|
|
63318
|
+
requests_per_minute: 30,
|
|
63319
|
+
max_concurrent: 2,
|
|
63320
|
+
acquire_timeout_ms: 10000,
|
|
63321
|
+
source: null,
|
|
63322
|
+
note: "Declared primary of the hot-path llm.fast alias but blocked on open item OI-01. Held tighter than the default because a hot-path caller cannot afford to queue: if the limit binds, failing fast into the chain beats waiting."
|
|
63323
|
+
},
|
|
63324
|
+
openrouter: {
|
|
63325
|
+
basis: "conservative-default",
|
|
63326
|
+
requests_per_minute: 30,
|
|
63327
|
+
max_concurrent: 2,
|
|
63328
|
+
acquire_timeout_ms: 15000,
|
|
63329
|
+
source: null,
|
|
63330
|
+
note: "Aggregator backstop, reached by explicit route only. Held tight because its own limits are a function of whichever upstream it selects, which the client cannot observe."
|
|
63331
|
+
}
|
|
63332
|
+
};
|
|
63333
|
+
var limitsConfig = {
|
|
63334
|
+
defaults: defaults$1,
|
|
63335
|
+
providers: providers$1
|
|
63336
|
+
};
|
|
63337
|
+
|
|
63338
|
+
/**
|
|
63339
|
+
* Client-side rate and concurrency guards, per provider (W4-04).
|
|
63340
|
+
*
|
|
63341
|
+
* A provider's rate limit is enforced at the provider whether or not the client
|
|
63342
|
+
* respects it. The reason to respect it here is what a 429 means once it
|
|
63343
|
+
* arrives: to the fallback chain it is indistinguishable from provider
|
|
63344
|
+
* ill-health, so a client that over-drives a healthy provider will open that
|
|
63345
|
+
* provider's circuit breaker, fail over to a more expensive leg, and keep doing
|
|
63346
|
+
* so — converting a self-inflicted pacing problem into a permanent routing
|
|
63347
|
+
* change nobody chose. Pacing at the client is what keeps the breaker measuring
|
|
63348
|
+
* the provider rather than measuring us.
|
|
63349
|
+
*
|
|
63350
|
+
* Two distinct bounds are applied because they fail differently. The rate bound
|
|
63351
|
+
* (requests per minute) protects the provider's published ceiling. The
|
|
63352
|
+
* concurrency bound protects the caller: a hundred simultaneous in-flight
|
|
63353
|
+
* requests will each wait behind the other ninety-nine at the provider, so
|
|
63354
|
+
* every one of them blows its latency budget and the fan-out produces a hundred
|
|
63355
|
+
* timeouts instead of a queue.
|
|
63356
|
+
*
|
|
63357
|
+
* Limits live in `provider-limits.json`, not here. A rate limit discovered
|
|
63358
|
+
* during an incident should be correctable by config, not by a release.
|
|
63359
|
+
*
|
|
63360
|
+
* @module llm/rate-guard
|
|
63361
|
+
*/
|
|
63362
|
+
/** Seconds in a minute, converting a published per-minute ceiling to a refill rate. */
|
|
63363
|
+
const SECONDS_PER_MINUTE = 60;
|
|
63364
|
+
/** One request consumes one token. */
|
|
63365
|
+
const TOKENS_PER_REQUEST = 1;
|
|
63366
|
+
const config$1 = limitsConfig;
|
|
63367
|
+
/**
|
|
63368
|
+
* Resolve the limits that apply to a provider.
|
|
63369
|
+
*
|
|
63370
|
+
* An unregistered provider falls back to the conservative defaults rather than
|
|
63371
|
+
* to no limit at all. Treating "unknown" as "unlimited" would make every newly
|
|
63372
|
+
* onboarded provider the one most likely to be over-driven, which is exactly
|
|
63373
|
+
* backwards: a new provider is the one whose real ceiling is least understood.
|
|
63374
|
+
*
|
|
63375
|
+
* @param provider The provider key.
|
|
63376
|
+
* @returns Its limits.
|
|
63377
|
+
*/
|
|
63378
|
+
function limitsFor(provider) {
|
|
63379
|
+
return config$1.providers[provider] ?? config$1.defaults;
|
|
63380
|
+
}
|
|
63381
|
+
/** Every provider with a recorded limit, plus whether it is published or a default. */
|
|
63382
|
+
function limitsInventory() {
|
|
63383
|
+
return Object.keys(config$1.providers)
|
|
63384
|
+
.sort()
|
|
63385
|
+
.map((provider) => ({ provider, limits: config$1.providers[provider] }));
|
|
63386
|
+
}
|
|
63387
|
+
/**
|
|
63388
|
+
* Thrown when a caller could not acquire a slot within its budget.
|
|
63389
|
+
*
|
|
63390
|
+
* Distinguished from a provider failure so the chain does not count it against
|
|
63391
|
+
* route health: the provider was never asked, so nothing was learned about it.
|
|
63392
|
+
*/
|
|
63393
|
+
class RateGuardTimeoutError extends Error {
|
|
63394
|
+
/** The provider whose guard could not admit the call. */
|
|
63395
|
+
provider;
|
|
63396
|
+
/** Which of the two bounds the caller waited on. */
|
|
63397
|
+
bound;
|
|
63398
|
+
/**
|
|
63399
|
+
* @param provider The provider.
|
|
63400
|
+
* @param bound Which bound was binding.
|
|
63401
|
+
* @param waitedMs How long the caller waited.
|
|
63402
|
+
*/
|
|
63403
|
+
constructor(provider, bound, waitedMs) {
|
|
63404
|
+
super(`client-side ${bound} guard for provider "${provider}" did not admit the call within ${waitedMs} ms. ` +
|
|
63405
|
+
"The provider was never contacted, so this says nothing about its health.");
|
|
63406
|
+
this.name = "RateGuardTimeoutError";
|
|
63407
|
+
this.provider = provider;
|
|
63408
|
+
this.bound = bound;
|
|
63409
|
+
}
|
|
63410
|
+
}
|
|
63411
|
+
/**
|
|
63412
|
+
* A counting semaphore bounding simultaneous in-flight calls.
|
|
63413
|
+
*
|
|
63414
|
+
* Written here rather than pulled from a dependency because it is fifteen lines
|
|
63415
|
+
* and because the waiting behaviour matters: a waiter that times out must be
|
|
63416
|
+
* removed from the queue, or a burst of abandoned callers permanently consumes
|
|
63417
|
+
* the permits that later callers need.
|
|
63418
|
+
*/
|
|
63419
|
+
class ConcurrencyGate {
|
|
63420
|
+
inFlight = 0;
|
|
63421
|
+
waiters = [];
|
|
63422
|
+
limit;
|
|
63423
|
+
provider;
|
|
63424
|
+
/**
|
|
63425
|
+
* @param provider The provider this gate guards.
|
|
63426
|
+
* @param limit Maximum simultaneous in-flight calls.
|
|
63427
|
+
*/
|
|
63428
|
+
constructor(provider, limit) {
|
|
63429
|
+
this.provider = provider;
|
|
63430
|
+
this.limit = limit;
|
|
63431
|
+
}
|
|
63432
|
+
/**
|
|
63433
|
+
* Wait for a permit.
|
|
63434
|
+
*
|
|
63435
|
+
* @param timeoutMs How long the caller is willing to queue.
|
|
63436
|
+
* @returns A release function the caller must invoke exactly once.
|
|
63437
|
+
*/
|
|
63438
|
+
async acquire(timeoutMs) {
|
|
63439
|
+
if (this.inFlight < this.limit) {
|
|
63440
|
+
this.inFlight += 1;
|
|
63441
|
+
return () => this.release();
|
|
63442
|
+
}
|
|
63443
|
+
await new Promise((resolve, reject) => {
|
|
63444
|
+
const timer = setTimeout(() => {
|
|
63445
|
+
const index = this.waiters.findIndex((waiter) => waiter.timer === timer);
|
|
63446
|
+
if (index !== -1) {
|
|
63447
|
+
this.waiters.splice(index, 1);
|
|
63448
|
+
}
|
|
63449
|
+
reject(new RateGuardTimeoutError(this.provider, "concurrency", timeoutMs));
|
|
63450
|
+
}, timeoutMs);
|
|
63451
|
+
this.waiters.push({ resolve, reject, timer });
|
|
63452
|
+
});
|
|
63453
|
+
this.inFlight += 1;
|
|
63454
|
+
return () => this.release();
|
|
63455
|
+
}
|
|
63456
|
+
/**
|
|
63457
|
+
* Return a permit and admit the next waiter.
|
|
63458
|
+
*
|
|
63459
|
+
* @returns void
|
|
63460
|
+
*/
|
|
63461
|
+
release() {
|
|
63462
|
+
this.inFlight -= 1;
|
|
63463
|
+
const next = this.waiters.shift();
|
|
63464
|
+
if (next !== undefined) {
|
|
63465
|
+
clearTimeout(next.timer);
|
|
63466
|
+
next.resolve();
|
|
63467
|
+
}
|
|
63468
|
+
}
|
|
63469
|
+
/**
|
|
63470
|
+
* @returns How many calls are currently in flight.
|
|
63471
|
+
*/
|
|
63472
|
+
inFlightCount() {
|
|
63473
|
+
return this.inFlight;
|
|
63474
|
+
}
|
|
63475
|
+
/**
|
|
63476
|
+
* @returns How many callers are queued.
|
|
63477
|
+
*/
|
|
63478
|
+
queueLength() {
|
|
63479
|
+
return this.waiters.length;
|
|
63480
|
+
}
|
|
63481
|
+
}
|
|
63482
|
+
/** Per-provider guards, created on first use and shared process-wide. */
|
|
63483
|
+
const rateLimiters = new Map();
|
|
63484
|
+
const concurrencyGates = new Map();
|
|
63485
|
+
/**
|
|
63486
|
+
* The rate limiter for a provider.
|
|
63487
|
+
*
|
|
63488
|
+
* Shared process-wide rather than per-call-site, because the provider's ceiling
|
|
63489
|
+
* applies to the process as a whole. Per-call-site limiters would each stay
|
|
63490
|
+
* under the ceiling while their sum sailed past it.
|
|
63491
|
+
*
|
|
63492
|
+
* @param provider The provider key.
|
|
63493
|
+
* @returns Its limiter.
|
|
63494
|
+
*/
|
|
63495
|
+
function rateLimiterFor(provider) {
|
|
63496
|
+
let limiter = rateLimiters.get(provider);
|
|
63497
|
+
if (limiter === undefined) {
|
|
63498
|
+
const limits = limitsFor(provider);
|
|
63499
|
+
limiter = new TokenBucketRateLimiter({
|
|
63500
|
+
maxTokens: limits.requests_per_minute,
|
|
63501
|
+
refillRate: limits.requests_per_minute / SECONDS_PER_MINUTE,
|
|
63502
|
+
label: `llm:${provider}`,
|
|
63503
|
+
timeoutMs: limits.acquire_timeout_ms,
|
|
63504
|
+
});
|
|
63505
|
+
rateLimiters.set(provider, limiter);
|
|
63506
|
+
}
|
|
63507
|
+
return limiter;
|
|
63508
|
+
}
|
|
63509
|
+
/**
|
|
63510
|
+
* The concurrency gate for a provider.
|
|
63511
|
+
*
|
|
63512
|
+
* @param provider The provider key.
|
|
63513
|
+
* @returns Its gate.
|
|
63514
|
+
*/
|
|
63515
|
+
function concurrencyGateFor(provider) {
|
|
63516
|
+
let gate = concurrencyGates.get(provider);
|
|
63517
|
+
if (gate === undefined) {
|
|
63518
|
+
gate = new ConcurrencyGate(provider, limitsFor(provider).max_concurrent);
|
|
63519
|
+
concurrencyGates.set(provider, gate);
|
|
63520
|
+
}
|
|
63521
|
+
return gate;
|
|
63522
|
+
}
|
|
63523
|
+
/**
|
|
63524
|
+
* Run a call under a provider's rate and concurrency guards.
|
|
63525
|
+
*
|
|
63526
|
+
* The concurrency permit is taken AFTER the rate token. Taking it first would
|
|
63527
|
+
* let callers hold scarce permits while idling in the rate queue, which
|
|
63528
|
+
* throttles the provider twice over and turns a pacing bound into a deadlock
|
|
63529
|
+
* shaped like slowness.
|
|
63530
|
+
*
|
|
63531
|
+
* `maxWaitMs` bounds how long a caller may queue. It exists because the queue
|
|
63532
|
+
* spends the SAME budget the call itself does: a caller that waits out its whole
|
|
63533
|
+
* deadline in a rate queue has failed just as completely as one that waited on
|
|
63534
|
+
* the provider, and worse, it never reached the fallback chain that could have
|
|
63535
|
+
* answered it. Passing the leg's own timeout keeps one clock governing the
|
|
63536
|
+
* whole attempt.
|
|
63537
|
+
*
|
|
63538
|
+
* @param provider The provider key.
|
|
63539
|
+
* @param call The work to run once admitted.
|
|
63540
|
+
* @param maxWaitMs Ceiling on queue time; the configured guard timeout applies when lower.
|
|
63541
|
+
* @returns The call's result.
|
|
63542
|
+
* @throws {RateGuardTimeoutError} When neither bound admitted the call in time.
|
|
63543
|
+
*/
|
|
63544
|
+
async function withProviderGuards(provider, call, maxWaitMs) {
|
|
63545
|
+
const limits = limitsFor(provider);
|
|
63546
|
+
const waitBudgetMs = maxWaitMs === undefined
|
|
63547
|
+
? limits.acquire_timeout_ms
|
|
63548
|
+
: Math.min(maxWaitMs, limits.acquire_timeout_ms);
|
|
63549
|
+
try {
|
|
63550
|
+
await rateLimiterFor(provider).acquire();
|
|
63551
|
+
}
|
|
63552
|
+
catch {
|
|
63553
|
+
throw new RateGuardTimeoutError(provider, "rate", waitBudgetMs);
|
|
63554
|
+
}
|
|
63555
|
+
const release = await concurrencyGateFor(provider).acquire(waitBudgetMs);
|
|
63556
|
+
try {
|
|
63557
|
+
return await call();
|
|
63558
|
+
}
|
|
63559
|
+
finally {
|
|
63560
|
+
// Released on every path. A permit leaked on the error path would shrink
|
|
63561
|
+
// the effective limit by one on each failure until nothing could run — and
|
|
63562
|
+
// failures cluster exactly when throughput matters most.
|
|
63563
|
+
release();
|
|
63564
|
+
}
|
|
63565
|
+
}
|
|
63566
|
+
/**
|
|
63567
|
+
* Inspect the guards currently in use.
|
|
63568
|
+
*
|
|
63569
|
+
* @returns A snapshot per provider that has been used, sorted by provider.
|
|
63570
|
+
*/
|
|
63571
|
+
function guardSnapshots() {
|
|
63572
|
+
const providers = new Set([...rateLimiters.keys(), ...concurrencyGates.keys()]);
|
|
63573
|
+
return [...providers].sort().map((provider) => {
|
|
63574
|
+
const limits = limitsFor(provider);
|
|
63575
|
+
const limiter = rateLimiters.get(provider);
|
|
63576
|
+
const gate = concurrencyGates.get(provider);
|
|
63577
|
+
return {
|
|
63578
|
+
provider,
|
|
63579
|
+
basis: limits.basis,
|
|
63580
|
+
requestsPerMinute: limits.requests_per_minute,
|
|
63581
|
+
maxConcurrent: limits.max_concurrent,
|
|
63582
|
+
inFlight: gate?.inFlightCount() ?? 0,
|
|
63583
|
+
rateQueueLength: limiter?.getQueueLength() ?? 0,
|
|
63584
|
+
concurrencyQueueLength: gate?.queueLength() ?? 0,
|
|
63585
|
+
availableTokens: limiter?.getAvailableTokens() ?? limits.requests_per_minute * TOKENS_PER_REQUEST,
|
|
63586
|
+
};
|
|
63587
|
+
});
|
|
63588
|
+
}
|
|
63589
|
+
/**
|
|
63590
|
+
* Discard all guard state.
|
|
63591
|
+
*
|
|
63592
|
+
* Exists so a test can start from a known position; a shared process-wide
|
|
63593
|
+
* limiter is otherwise carried between tests and makes their order matter.
|
|
63594
|
+
*
|
|
63595
|
+
* @returns void
|
|
63596
|
+
*/
|
|
63597
|
+
function resetProviderGuards() {
|
|
63598
|
+
for (const limiter of rateLimiters.values()) {
|
|
63599
|
+
limiter.reset();
|
|
63600
|
+
}
|
|
63601
|
+
rateLimiters.clear();
|
|
63602
|
+
concurrencyGates.clear();
|
|
63603
|
+
}
|
|
63604
|
+
|
|
63605
|
+
/**
|
|
63606
|
+
* Ordered execution of an alias's fallback chain (PD-3).
|
|
63607
|
+
*
|
|
63608
|
+
* Every call walks the chain primary -> secondary -> closed incumbent, and each
|
|
63609
|
+
* leg runs under a hard timeout and a circuit breaker. Those three controls are
|
|
63610
|
+
* one mechanism rather than three features: a chain without timeouts never
|
|
63611
|
+
* reaches its second leg, a chain without a breaker pays a dead provider's full
|
|
63612
|
+
* timeout on every call, and a timeout without a chain is just a slower
|
|
63613
|
+
* failure. Timeout cascades are this system's known brown-out mode, which is
|
|
63614
|
+
* why the budget is enforced here — at the only place that knows both the
|
|
63615
|
+
* caller's deadline and how many legs are left to spend it on.
|
|
63616
|
+
*
|
|
63617
|
+
* Nothing here ever substitutes a value for an outcome. When every leg is
|
|
63618
|
+
* exhausted the caller gets a typed error naming each leg and why it failed,
|
|
63619
|
+
* because a default returned in place of an answer is a wrong answer that
|
|
63620
|
+
* nobody is told about.
|
|
63621
|
+
*
|
|
63622
|
+
* @module llm/fallback-chain
|
|
63623
|
+
*/
|
|
63624
|
+
/** Zero-valued usage, used as the identity when summing across attempts. */
|
|
63625
|
+
const EMPTY_USAGE = {
|
|
63626
|
+
prompt_tokens: 0,
|
|
63627
|
+
completion_tokens: 0,
|
|
63628
|
+
provider: "none",
|
|
63629
|
+
model: "none",
|
|
63630
|
+
cost: 0,
|
|
63631
|
+
};
|
|
63632
|
+
/**
|
|
63633
|
+
* Thrown when every leg of a chain has been tried and none produced an answer.
|
|
63634
|
+
*
|
|
63635
|
+
* Carries the full attempt record rather than only the last error. The last
|
|
63636
|
+
* error is usually the least informative one — the incumbent timing out says
|
|
63637
|
+
* nothing about why the two legs before it were skipped — and an operator
|
|
63638
|
+
* reading only that would go looking in the wrong place.
|
|
63639
|
+
*/
|
|
63640
|
+
class ChainExhaustedError extends Error {
|
|
63641
|
+
/** The alias whose chain was exhausted. */
|
|
63642
|
+
alias;
|
|
63643
|
+
/** Every leg tried, in order, with its outcome. */
|
|
63644
|
+
attempts;
|
|
63645
|
+
/** Usage spent across the failed attempts, so the spend is still accounted for. */
|
|
63646
|
+
totalUsage;
|
|
63647
|
+
/**
|
|
63648
|
+
* @param alias The alias.
|
|
63649
|
+
* @param attempts The attempt record.
|
|
63650
|
+
* @param totalUsage Usage spent across all attempts.
|
|
63651
|
+
*/
|
|
63652
|
+
constructor(alias, attempts, totalUsage) {
|
|
63653
|
+
const detail = attempts
|
|
63654
|
+
.map((attempt) => `${attempt.role}(${attempt.provider}/${attempt.modelId}): ${attempt.outcome}` +
|
|
63655
|
+
(attempt.reason === undefined ? "" : ` — ${attempt.reason}`))
|
|
63656
|
+
.join("; ");
|
|
63657
|
+
super(`LLM alias "${alias}" exhausted its fallback chain. Attempts: ${detail || "(no leg was servable)"}`);
|
|
63658
|
+
this.name = "ChainExhaustedError";
|
|
63659
|
+
this.alias = alias;
|
|
63660
|
+
this.attempts = attempts;
|
|
63661
|
+
this.totalUsage = totalUsage;
|
|
63662
|
+
}
|
|
63663
|
+
}
|
|
63664
|
+
/**
|
|
63665
|
+
* Add two usage records.
|
|
63666
|
+
*
|
|
63667
|
+
* Attribution keeps the LAST attempt's provider and model, because that is the
|
|
63668
|
+
* one that produced the answer the caller is holding, while the token counts
|
|
63669
|
+
* accumulate across every attempt. Charging only the successful attempt would
|
|
63670
|
+
* understate spend by exactly the amount the failures cost — which is the
|
|
63671
|
+
* amount a fallback chain is most likely to run up.
|
|
63672
|
+
*
|
|
63673
|
+
* @param a The running total.
|
|
63674
|
+
* @param b The attempt to add, if any.
|
|
63675
|
+
* @returns The combined usage.
|
|
63676
|
+
*/
|
|
63677
|
+
function sumUsage(a, b) {
|
|
63678
|
+
if (b === undefined) {
|
|
63679
|
+
return a;
|
|
63680
|
+
}
|
|
63681
|
+
return {
|
|
63682
|
+
prompt_tokens: a.prompt_tokens + b.prompt_tokens,
|
|
63683
|
+
completion_tokens: a.completion_tokens + b.completion_tokens,
|
|
63684
|
+
reasoning_tokens: a.reasoning_tokens === undefined && b.reasoning_tokens === undefined
|
|
63685
|
+
? undefined
|
|
63686
|
+
: (a.reasoning_tokens ?? 0) + (b.reasoning_tokens ?? 0),
|
|
63687
|
+
cached_tokens: a.cached_tokens === undefined && b.cached_tokens === undefined
|
|
63688
|
+
? undefined
|
|
63689
|
+
: (a.cached_tokens ?? 0) + (b.cached_tokens ?? 0),
|
|
63690
|
+
provider: b.provider,
|
|
63691
|
+
model: b.model,
|
|
63692
|
+
cost: a.cost + b.cost,
|
|
63693
|
+
};
|
|
63694
|
+
}
|
|
63695
|
+
/** Raised internally when a leg exceeds its budget. */
|
|
63696
|
+
class LegTimeoutError extends Error {
|
|
63697
|
+
/**
|
|
63698
|
+
* @param routeKey The leg that timed out.
|
|
63699
|
+
* @param budgetMs Its budget in milliseconds.
|
|
63700
|
+
*/
|
|
63701
|
+
constructor(routeKey, budgetMs) {
|
|
63702
|
+
super(`route ${routeKey} exceeded its ${budgetMs} ms budget`);
|
|
63703
|
+
this.name = "LegTimeoutError";
|
|
63704
|
+
}
|
|
63705
|
+
}
|
|
63706
|
+
/**
|
|
63707
|
+
* Run one leg under a hard timeout, honouring the caller's own cancellation.
|
|
63708
|
+
*
|
|
63709
|
+
* The timer is always cleared and the abort listener always removed, including
|
|
63710
|
+
* on the success path. A long-lived process that leaked one timer per LLM call
|
|
63711
|
+
* would accumulate them at exactly the rate it does useful work.
|
|
63712
|
+
*
|
|
63713
|
+
* @param leg The leg to run.
|
|
63714
|
+
* @param params Normalised parameters for this leg.
|
|
63715
|
+
* @param execution The call context.
|
|
63716
|
+
* @returns The provider's answer.
|
|
63717
|
+
*/
|
|
63718
|
+
async function runLeg(leg, params, execution) {
|
|
63719
|
+
const controller = new AbortController();
|
|
63720
|
+
const budgetMs = leg.route.timeoutMs;
|
|
63721
|
+
const timer = setTimeout(() => {
|
|
63722
|
+
controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
|
|
63723
|
+
}, budgetMs);
|
|
63724
|
+
const forwardAbort = () => {
|
|
63725
|
+
controller.abort(execution.callerSignal?.reason);
|
|
63726
|
+
};
|
|
63727
|
+
if (execution.callerSignal !== undefined) {
|
|
63728
|
+
if (execution.callerSignal.aborted) {
|
|
63729
|
+
forwardAbort();
|
|
63730
|
+
}
|
|
63731
|
+
else {
|
|
63732
|
+
execution.callerSignal.addEventListener("abort", forwardAbort, { once: true });
|
|
63733
|
+
}
|
|
63734
|
+
}
|
|
63735
|
+
try {
|
|
63736
|
+
// The guards wrap the transport rather than the whole leg, so the per-leg
|
|
63737
|
+
// timeout above still bounds the total wait: a caller queued behind the
|
|
63738
|
+
// rate limiter is spending its budget just as surely as one waiting on the
|
|
63739
|
+
// provider, and only one clock should govern both.
|
|
63740
|
+
return await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
|
|
63741
|
+
route: leg.route,
|
|
63742
|
+
content: execution.content,
|
|
63743
|
+
responseFormat: execution.responseFormat,
|
|
63744
|
+
params,
|
|
63745
|
+
signal: controller.signal,
|
|
63746
|
+
correlationId: execution.correlationId,
|
|
63747
|
+
}), budgetMs);
|
|
63748
|
+
}
|
|
63749
|
+
finally {
|
|
63750
|
+
clearTimeout(timer);
|
|
63751
|
+
execution.callerSignal?.removeEventListener("abort", forwardAbort);
|
|
63752
|
+
}
|
|
63753
|
+
}
|
|
63754
|
+
/**
|
|
63755
|
+
* Classify why a leg failed.
|
|
63756
|
+
*
|
|
63757
|
+
* The distinction matters to the breaker: a timeout and a 5xx are evidence the
|
|
63758
|
+
* provider is unhealthy, while the caller cancelling is not. Counting a
|
|
63759
|
+
* cancellation as a provider failure would let a burst of user-cancelled
|
|
63760
|
+
* requests open the breaker on a perfectly healthy route.
|
|
63761
|
+
*
|
|
63762
|
+
* @param error The thrown value.
|
|
63763
|
+
* @param callerSignal The caller's cancellation signal, if any.
|
|
63764
|
+
* @returns The outcome and whether it counts against route health.
|
|
63765
|
+
*/
|
|
63766
|
+
function classify(error, callerSignal) {
|
|
63767
|
+
if (callerSignal !== undefined && callerSignal.aborted) {
|
|
63768
|
+
return {
|
|
63769
|
+
outcome: "skipped",
|
|
63770
|
+
reason: "caller cancelled",
|
|
63771
|
+
countsAgainstHealth: false,
|
|
63772
|
+
};
|
|
63773
|
+
}
|
|
63774
|
+
if (error instanceof LegTimeoutError) {
|
|
63775
|
+
return { outcome: "timeout", reason: error.message, countsAgainstHealth: true };
|
|
63776
|
+
}
|
|
63777
|
+
if (error instanceof UnsupportedCapabilityError) {
|
|
63778
|
+
return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
|
|
63779
|
+
}
|
|
63780
|
+
if (error instanceof RateGuardTimeoutError) {
|
|
63781
|
+
// Self-inflicted pacing, not provider ill-health. Counting it would let the
|
|
63782
|
+
// client's own throttling open a breaker on a perfectly healthy provider
|
|
63783
|
+
// and permanently reroute traffic nobody chose to reroute.
|
|
63784
|
+
return { outcome: "skipped", reason: error.message, countsAgainstHealth: false };
|
|
63785
|
+
}
|
|
63786
|
+
const reason = error instanceof Error ? error.message : String(error);
|
|
63787
|
+
if (/abort/i.test(reason)) {
|
|
63788
|
+
return {
|
|
63789
|
+
outcome: "timeout",
|
|
63790
|
+
reason: `aborted: ${reason}`,
|
|
63791
|
+
countsAgainstHealth: true,
|
|
63792
|
+
};
|
|
63793
|
+
}
|
|
63794
|
+
return { outcome: "error", reason, countsAgainstHealth: true };
|
|
63795
|
+
}
|
|
63796
|
+
/**
|
|
63797
|
+
* Whether the caller has stopped waiting.
|
|
63798
|
+
*
|
|
63799
|
+
* Read through a function rather than inline, because `AbortSignal.aborted` is
|
|
63800
|
+
* a live getter: it can flip to true while a leg is in flight, but a compiler
|
|
63801
|
+
* that narrowed it at the top of the loop would prove the later check
|
|
63802
|
+
* unreachable and invite its removal. The check is not redundant — it is the
|
|
63803
|
+
* only thing that stops the chain spending money on an answer nobody will read.
|
|
63804
|
+
*
|
|
63805
|
+
* @param signal The caller's signal, if any.
|
|
63806
|
+
* @returns Whether the call has been cancelled.
|
|
63807
|
+
*/
|
|
63808
|
+
function isAborted$1(signal) {
|
|
63809
|
+
return signal !== undefined && signal.aborted;
|
|
63810
|
+
}
|
|
63811
|
+
/**
|
|
63812
|
+
* Walk a chain until a leg answers.
|
|
63813
|
+
*
|
|
63814
|
+
* @param alias The alias being served, for error attribution.
|
|
63815
|
+
* @param execution The call context.
|
|
63816
|
+
* @returns The first successful leg's answer, with the full attempt record.
|
|
63817
|
+
* @throws {ChainExhaustedError} When no leg produced an answer.
|
|
63818
|
+
*/
|
|
63819
|
+
async function executeChain(alias, execution) {
|
|
63820
|
+
const now = execution.now ?? Date.now;
|
|
63821
|
+
const attempts = [];
|
|
63822
|
+
let totalUsage = EMPTY_USAGE;
|
|
63823
|
+
for (const leg of execution.legs) {
|
|
63824
|
+
const { route } = leg;
|
|
63825
|
+
if (isAborted$1(execution.callerSignal)) {
|
|
63826
|
+
// The caller has stopped waiting. Continuing to walk the chain would
|
|
63827
|
+
// spend money on an answer nobody will read.
|
|
63828
|
+
break;
|
|
63829
|
+
}
|
|
63830
|
+
if (leg.params instanceof UnsupportedCapabilityError) {
|
|
63831
|
+
const record = {
|
|
63832
|
+
routeKey: route.routeKey,
|
|
63833
|
+
role: route.role,
|
|
63834
|
+
provider: route.providerName,
|
|
63835
|
+
modelId: route.modelId,
|
|
63836
|
+
outcome: "skipped",
|
|
63837
|
+
durationMs: 0,
|
|
63838
|
+
reason: leg.params.message,
|
|
63839
|
+
};
|
|
63840
|
+
attempts.push(record);
|
|
63841
|
+
execution.onAttempt?.(record);
|
|
63842
|
+
continue;
|
|
63843
|
+
}
|
|
63844
|
+
if (!execution.breakers.allows(route.routeKey)) {
|
|
63845
|
+
const record = {
|
|
63846
|
+
routeKey: route.routeKey,
|
|
63847
|
+
role: route.role,
|
|
63848
|
+
provider: route.providerName,
|
|
63849
|
+
modelId: route.modelId,
|
|
63850
|
+
outcome: "breaker-open",
|
|
63851
|
+
durationMs: 0,
|
|
63852
|
+
reason: `circuit breaker is ${execution.breakers.stateOf(route.routeKey)}`,
|
|
63853
|
+
};
|
|
63854
|
+
attempts.push(record);
|
|
63855
|
+
execution.onAttempt?.(record);
|
|
63856
|
+
continue;
|
|
63857
|
+
}
|
|
63858
|
+
const startedAt = now();
|
|
63859
|
+
execution.breakers.onAttemptStart(route.routeKey);
|
|
63860
|
+
try {
|
|
63861
|
+
const response = await runLeg(leg, leg.params, execution);
|
|
63862
|
+
execution.breakers.onSuccess(route.routeKey);
|
|
63863
|
+
totalUsage = sumUsage(totalUsage, response.usage);
|
|
63864
|
+
const record = {
|
|
63865
|
+
routeKey: route.routeKey,
|
|
63866
|
+
role: route.role,
|
|
63867
|
+
provider: route.providerName,
|
|
63868
|
+
modelId: route.modelId,
|
|
63869
|
+
outcome: "ok",
|
|
63870
|
+
durationMs: now() - startedAt,
|
|
63871
|
+
usage: response.usage,
|
|
63872
|
+
};
|
|
63873
|
+
attempts.push(record);
|
|
63874
|
+
execution.onAttempt?.(record);
|
|
63875
|
+
return { response, servedBy: route, attempts, totalUsage };
|
|
63876
|
+
}
|
|
63877
|
+
catch (error) {
|
|
63878
|
+
const { outcome, reason, countsAgainstHealth } = classify(error, execution.callerSignal);
|
|
63879
|
+
if (countsAgainstHealth) {
|
|
63880
|
+
execution.breakers.onFailure(route.routeKey);
|
|
63881
|
+
}
|
|
63882
|
+
const record = {
|
|
63883
|
+
routeKey: route.routeKey,
|
|
63884
|
+
role: route.role,
|
|
63885
|
+
provider: route.providerName,
|
|
63886
|
+
modelId: route.modelId,
|
|
63887
|
+
outcome,
|
|
63888
|
+
durationMs: now() - startedAt,
|
|
63889
|
+
reason,
|
|
63890
|
+
};
|
|
63891
|
+
attempts.push(record);
|
|
63892
|
+
execution.onAttempt?.(record);
|
|
63893
|
+
if (outcome === "skipped" && isAborted$1(execution.callerSignal)) {
|
|
63894
|
+
break;
|
|
63895
|
+
}
|
|
63896
|
+
}
|
|
63897
|
+
}
|
|
63898
|
+
throw new ChainExhaustedError(alias, attempts, totalUsage);
|
|
63899
|
+
}
|
|
63900
|
+
|
|
63901
|
+
var schema_version = 1;
|
|
63902
|
+
var policy_source = "docs/llm-provider-migration.md#3-routing-policy (v1.1)";
|
|
63903
|
+
var revised = "2026-09-10";
|
|
63904
|
+
var defaults = {
|
|
63905
|
+
request_timeout_ms: {
|
|
63906
|
+
"hot-path": 30000,
|
|
63907
|
+
background: 90000,
|
|
63908
|
+
batch: 300000
|
|
63909
|
+
},
|
|
63910
|
+
retries_per_leg: 1,
|
|
63911
|
+
circuit_breaker: {
|
|
63912
|
+
failure_threshold: 5,
|
|
63913
|
+
cooldown_ms: 60000,
|
|
63914
|
+
half_open_probes: 1
|
|
63915
|
+
}
|
|
63916
|
+
};
|
|
63917
|
+
var providers = {
|
|
63918
|
+
anthropic: {
|
|
63919
|
+
display_name: "Anthropic",
|
|
63920
|
+
tier: "closed",
|
|
63921
|
+
api_style: "anthropic",
|
|
63922
|
+
base_url: null,
|
|
63923
|
+
base_url_env: null,
|
|
63924
|
+
api_key_env: "ANTHROPIC_API_KEY",
|
|
63925
|
+
secret_path: "llm/anthropic/apiKey",
|
|
63926
|
+
account_status: "live",
|
|
63927
|
+
lumic_provider: "anthropic",
|
|
63928
|
+
published_rpm: null,
|
|
63929
|
+
published_tpm: null,
|
|
63930
|
+
limits_source: null,
|
|
63931
|
+
docs_url: "https://docs.claude.com/en/api/overview",
|
|
63932
|
+
notes: "Permanent designated tier. Retained, keyed, budgeted and continuously exercised as the revert target of every alias (PD-11)."
|
|
63933
|
+
},
|
|
63934
|
+
openai: {
|
|
63935
|
+
display_name: "OpenAI",
|
|
63936
|
+
tier: "closed",
|
|
63937
|
+
api_style: "openai-compatible",
|
|
63938
|
+
base_url: null,
|
|
63939
|
+
base_url_env: null,
|
|
63940
|
+
api_key_env: "OPENAI_API_KEY",
|
|
63941
|
+
secret_path: "llm/openai/apiKey",
|
|
63942
|
+
account_status: "live",
|
|
63943
|
+
lumic_provider: "openai",
|
|
63944
|
+
published_rpm: null,
|
|
63945
|
+
published_tpm: null,
|
|
63946
|
+
limits_source: null,
|
|
63947
|
+
docs_url: "https://platform.openai.com/docs/api-reference",
|
|
63948
|
+
notes: "Permanent designated tier. No alias routes to it today; it stays keyed and budgeted so a per-alias revert to an OpenAI incumbent is a config change (PD-11)."
|
|
63949
|
+
},
|
|
63950
|
+
deepinfra: {
|
|
63951
|
+
display_name: "DeepInfra",
|
|
63952
|
+
tier: "open",
|
|
63953
|
+
api_style: "openai-compatible",
|
|
63954
|
+
base_url: "https://api.deepinfra.com/v1/openai",
|
|
63955
|
+
base_url_env: "DEEPINFRA_BASE_URL",
|
|
63956
|
+
api_key_env: "DEEPINFRA_API_KEY",
|
|
63957
|
+
secret_path: "llm/deepinfra/apiKey",
|
|
63958
|
+
account_status: "pending-onboarding",
|
|
63959
|
+
lumic_provider: null,
|
|
63960
|
+
published_rpm: null,
|
|
63961
|
+
published_tpm: null,
|
|
63962
|
+
limits_source: null,
|
|
63963
|
+
docs_url: "https://deepinfra.com/docs",
|
|
63964
|
+
notes: "Rate limits and exact model ids are transcribed from the provider console at W3-02, never guessed."
|
|
63965
|
+
},
|
|
63966
|
+
fireworks: {
|
|
63967
|
+
display_name: "Fireworks AI",
|
|
63968
|
+
tier: "open",
|
|
63969
|
+
api_style: "openai-compatible",
|
|
63970
|
+
base_url: "https://api.fireworks.ai/inference/v1",
|
|
63971
|
+
base_url_env: "FIREWORKS_BASE_URL",
|
|
63972
|
+
api_key_env: "FIREWORKS_API_KEY",
|
|
63973
|
+
secret_path: "llm/fireworks/apiKey",
|
|
63974
|
+
account_status: "pending-onboarding",
|
|
63975
|
+
lumic_provider: null,
|
|
63976
|
+
published_rpm: null,
|
|
63977
|
+
published_tpm: null,
|
|
63978
|
+
limits_source: null,
|
|
63979
|
+
docs_url: "https://docs.fireworks.ai",
|
|
63980
|
+
notes: "Onboarded as a capacity backstop. No alias routes to it until its published RPM ceiling is recorded in provider-limits (W4-04)."
|
|
63981
|
+
},
|
|
63982
|
+
zai: {
|
|
63983
|
+
display_name: "Z.ai",
|
|
63984
|
+
tier: "open",
|
|
63985
|
+
api_style: "openai-compatible",
|
|
63986
|
+
base_url: "https://api.z.ai/api/paas/v4",
|
|
63987
|
+
base_url_env: "ZAI_BASE_URL",
|
|
63988
|
+
api_key_env: "ZAI_API_KEY",
|
|
63989
|
+
secret_path: "llm/zai/apiKey",
|
|
63990
|
+
account_status: "pending-onboarding",
|
|
63991
|
+
lumic_provider: null,
|
|
63992
|
+
published_rpm: null,
|
|
63993
|
+
published_tpm: null,
|
|
63994
|
+
limits_source: null,
|
|
63995
|
+
docs_url: "https://docs.z.ai",
|
|
63996
|
+
notes: "First-party GLM host. Section 3 anchors first-party GLM below the converged multi-host price, so it is the primary leg for llm.reason."
|
|
63997
|
+
},
|
|
63998
|
+
deepseek: {
|
|
63999
|
+
display_name: "DeepSeek",
|
|
64000
|
+
tier: "open",
|
|
64001
|
+
api_style: "openai-compatible",
|
|
64002
|
+
base_url: "https://api.deepseek.com",
|
|
64003
|
+
base_url_env: "DEEPSEEK_BASE_URL",
|
|
64004
|
+
api_key_env: "DEEPSEEK_API_KEY",
|
|
64005
|
+
secret_path: "llm/deepseek/apiKey",
|
|
64006
|
+
account_status: "live",
|
|
64007
|
+
lumic_provider: "deepseek",
|
|
64008
|
+
published_rpm: null,
|
|
64009
|
+
published_tpm: null,
|
|
64010
|
+
limits_source: null,
|
|
64011
|
+
docs_url: "https://api-docs.deepseek.com",
|
|
64012
|
+
notes: "Already a registered lumic provider, so its legs can also be served by the degraded direct transport. Off-peak windows are captured in gateway config at W3-05 for batch scheduling."
|
|
64013
|
+
},
|
|
64014
|
+
groq: {
|
|
64015
|
+
display_name: "Groq",
|
|
64016
|
+
tier: "open",
|
|
64017
|
+
api_style: "openai-compatible",
|
|
64018
|
+
base_url: "https://api.groq.com/openai/v1",
|
|
64019
|
+
base_url_env: "GROQ_BASE_URL",
|
|
64020
|
+
api_key_env: "GROQ_API_KEY",
|
|
64021
|
+
secret_path: "llm/groq/apiKey",
|
|
64022
|
+
account_status: "not-in-scope",
|
|
64023
|
+
lumic_provider: null,
|
|
64024
|
+
published_rpm: null,
|
|
64025
|
+
published_tpm: null,
|
|
64026
|
+
limits_source: null,
|
|
64027
|
+
docs_url: "https://console.groq.com/docs",
|
|
64028
|
+
notes: "Section 3 names Groq or Cerebras as the llm.fast primary, but the W3 onboarding list does not include either. See open item OI-01: until an owner resolves it, llm.fast's primary leg is unreachable and the alias serves from its secondary and closed legs."
|
|
64029
|
+
},
|
|
64030
|
+
openrouter: {
|
|
64031
|
+
display_name: "OpenRouter",
|
|
64032
|
+
tier: "aggregator",
|
|
64033
|
+
api_style: "openai-compatible",
|
|
64034
|
+
base_url: "https://openrouter.ai/api/v1",
|
|
64035
|
+
base_url_env: "OPENROUTER_BASE_URL",
|
|
64036
|
+
api_key_env: "OPENROUTER_API_KEY",
|
|
64037
|
+
secret_path: "llm/openrouter/apiKey",
|
|
64038
|
+
account_status: "pending-onboarding",
|
|
64039
|
+
lumic_provider: null,
|
|
64040
|
+
published_rpm: null,
|
|
64041
|
+
published_tpm: null,
|
|
64042
|
+
limits_source: null,
|
|
64043
|
+
docs_url: "https://openrouter.ai/docs",
|
|
64044
|
+
notes: "Aggregator backstop. Reaches an open-weight model when its dedicated host is down, without a new account. Not a chain leg by default: an aggregator hides which upstream served a call, which defeats per-provider attribution."
|
|
64045
|
+
}
|
|
64046
|
+
};
|
|
64047
|
+
var aliases = {
|
|
64048
|
+
"llm.reason": {
|
|
64049
|
+
workload: "Deep reasoning, audit loops, config tuning",
|
|
64050
|
+
latency_class: "background",
|
|
64051
|
+
criticality: "ops",
|
|
64052
|
+
isolation_capable: true,
|
|
64053
|
+
eval_gate: "free-text-judge",
|
|
64054
|
+
budget: {
|
|
64055
|
+
basis: "provisional-pre-baseline",
|
|
64056
|
+
monthly_usd: 750,
|
|
64057
|
+
alert_pct: [
|
|
64058
|
+
50,
|
|
64059
|
+
80,
|
|
64060
|
+
100
|
|
64061
|
+
]
|
|
64062
|
+
},
|
|
64063
|
+
routes: [
|
|
64064
|
+
{
|
|
64065
|
+
role: "primary",
|
|
64066
|
+
provider: "zai",
|
|
64067
|
+
model_id: null,
|
|
64068
|
+
model_id_status: "pending-provider-confirmation",
|
|
64069
|
+
model_id_source: null,
|
|
64070
|
+
model_family: "GLM-5.3",
|
|
64071
|
+
lumic_model: null,
|
|
64072
|
+
params: {
|
|
64073
|
+
temperature: null,
|
|
64074
|
+
max_output_tokens: null,
|
|
64075
|
+
supports_temperature: true,
|
|
64076
|
+
supports_json_schema: true,
|
|
64077
|
+
supports_tools: true,
|
|
64078
|
+
supports_cache_control: false,
|
|
64079
|
+
supports_vision: false
|
|
64080
|
+
},
|
|
64081
|
+
price_per_mtok: {
|
|
64082
|
+
input: 1.09,
|
|
64083
|
+
output: 3.43,
|
|
64084
|
+
as_of: "2026-09-10",
|
|
64085
|
+
source: "docs/llm-provider-migration.md#3 first-party GLM-5.3 anchor; re-verify at W2-02"
|
|
64086
|
+
}
|
|
64087
|
+
},
|
|
64088
|
+
{
|
|
64089
|
+
role: "secondary",
|
|
64090
|
+
provider: "deepinfra",
|
|
64091
|
+
model_id: null,
|
|
64092
|
+
model_id_status: "pending-provider-confirmation",
|
|
64093
|
+
model_id_source: null,
|
|
64094
|
+
model_family: "DeepSeek V4 Pro",
|
|
64095
|
+
lumic_model: null,
|
|
64096
|
+
params: {
|
|
64097
|
+
temperature: null,
|
|
64098
|
+
max_output_tokens: null,
|
|
64099
|
+
supports_temperature: true,
|
|
64100
|
+
supports_json_schema: true,
|
|
64101
|
+
supports_tools: true,
|
|
64102
|
+
supports_cache_control: false,
|
|
64103
|
+
supports_vision: false
|
|
64104
|
+
},
|
|
64105
|
+
price_per_mtok: {
|
|
64106
|
+
input: 1.3,
|
|
64107
|
+
output: 2.6,
|
|
64108
|
+
as_of: "2026-09-10",
|
|
64109
|
+
source: "docs/llm-provider-migration.md#3 DeepInfra anchor; re-verify at W2-02"
|
|
64110
|
+
}
|
|
64111
|
+
},
|
|
64112
|
+
{
|
|
64113
|
+
role: "closed_incumbent",
|
|
64114
|
+
provider: "anthropic",
|
|
64115
|
+
model_id: "claude-opus-4-7",
|
|
64116
|
+
model_id_status: "confirmed",
|
|
64117
|
+
model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
|
|
64118
|
+
model_family: "Opus-class",
|
|
64119
|
+
lumic_model: "claude-opus-4-7",
|
|
64120
|
+
params: {
|
|
64121
|
+
temperature: null,
|
|
64122
|
+
max_output_tokens: 128000,
|
|
64123
|
+
supports_temperature: true,
|
|
64124
|
+
supports_json_schema: true,
|
|
64125
|
+
supports_tools: true,
|
|
64126
|
+
supports_cache_control: true,
|
|
64127
|
+
supports_vision: true,
|
|
64128
|
+
context_window: 1000000
|
|
64129
|
+
},
|
|
64130
|
+
price_per_mtok: {
|
|
64131
|
+
input: 5,
|
|
64132
|
+
output: 25,
|
|
64133
|
+
as_of: "2026-09-10",
|
|
64134
|
+
source: "docs/llm-provider-migration.md#3 Opus-class anchor; matches lumic-utils/src/functions/llm-config.ts"
|
|
64135
|
+
}
|
|
64136
|
+
}
|
|
64137
|
+
]
|
|
64138
|
+
},
|
|
64139
|
+
"llm.agentic": {
|
|
64140
|
+
workload: "Hardest long-horizon agentic work",
|
|
64141
|
+
latency_class: "background",
|
|
64142
|
+
criticality: "ops",
|
|
64143
|
+
isolation_capable: true,
|
|
64144
|
+
eval_gate: "tool-call",
|
|
64145
|
+
policy_note: "Section 3 names the primary as Fable-class. The strongest Anthropic model registered in the lumic model registry — the table this codebase actually routes on — is claude-opus-4-7, so that is the configured id. Registering a Fable-class model in @adaptic/lumic-utils would change live model selection for every consumer of the advanced tier and is therefore a separate, evidence-gated decision, recorded as open item OI-02 rather than smuggled in here.",
|
|
64146
|
+
budget: {
|
|
64147
|
+
basis: "provisional-pre-baseline",
|
|
64148
|
+
monthly_usd: 1500,
|
|
64149
|
+
alert_pct: [
|
|
64150
|
+
50,
|
|
64151
|
+
80,
|
|
64152
|
+
100
|
|
64153
|
+
]
|
|
64154
|
+
},
|
|
64155
|
+
routes: [
|
|
64156
|
+
{
|
|
64157
|
+
role: "primary",
|
|
64158
|
+
provider: "anthropic",
|
|
64159
|
+
model_id: "claude-opus-4-7",
|
|
64160
|
+
model_id_status: "confirmed",
|
|
64161
|
+
model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
|
|
64162
|
+
model_family: "Fable-class (see policy_note)",
|
|
64163
|
+
lumic_model: "claude-opus-4-7",
|
|
64164
|
+
params: {
|
|
64165
|
+
temperature: null,
|
|
64166
|
+
max_output_tokens: 128000,
|
|
64167
|
+
supports_temperature: true,
|
|
64168
|
+
supports_json_schema: true,
|
|
64169
|
+
supports_tools: true,
|
|
64170
|
+
supports_cache_control: true,
|
|
64171
|
+
supports_vision: true,
|
|
64172
|
+
context_window: 1000000
|
|
64173
|
+
},
|
|
64174
|
+
price_per_mtok: {
|
|
64175
|
+
input: 10,
|
|
64176
|
+
output: 50,
|
|
64177
|
+
as_of: "2026-09-10",
|
|
64178
|
+
source: "docs/llm-provider-migration.md#3 Fable-class anchor"
|
|
64179
|
+
}
|
|
64180
|
+
},
|
|
64181
|
+
{
|
|
64182
|
+
role: "secondary",
|
|
64183
|
+
provider: "deepinfra",
|
|
64184
|
+
model_id: null,
|
|
64185
|
+
model_id_status: "pending-provider-confirmation",
|
|
64186
|
+
model_id_source: null,
|
|
64187
|
+
model_family: "Kimi K3",
|
|
64188
|
+
lumic_model: null,
|
|
64189
|
+
shadow_only: true,
|
|
64190
|
+
params: {
|
|
64191
|
+
temperature: null,
|
|
64192
|
+
max_output_tokens: null,
|
|
64193
|
+
supports_temperature: true,
|
|
64194
|
+
supports_json_schema: true,
|
|
64195
|
+
supports_tools: true,
|
|
64196
|
+
supports_cache_control: false,
|
|
64197
|
+
supports_vision: false
|
|
64198
|
+
},
|
|
64199
|
+
price_per_mtok: {
|
|
64200
|
+
input: 2.85,
|
|
64201
|
+
output: 14.25,
|
|
64202
|
+
as_of: "2026-09-10",
|
|
64203
|
+
source: "docs/llm-provider-migration.md#3 DeepInfra anchor; re-verify at W2-02"
|
|
64204
|
+
},
|
|
64205
|
+
notes: "Shadow-eval only per Section 3. Configured and scored, never served to a caller, so the hardest agentic path keeps its closed primary while the open candidate accrues evidence."
|
|
64206
|
+
}
|
|
64207
|
+
]
|
|
64208
|
+
},
|
|
64209
|
+
"llm.fast": {
|
|
64210
|
+
workload: "Latency-critical paths",
|
|
64211
|
+
latency_class: "hot-path",
|
|
64212
|
+
criticality: "trading-adjacent",
|
|
64213
|
+
isolation_capable: false,
|
|
64214
|
+
eval_gate: "structured-output",
|
|
64215
|
+
policy_note: "Trading-adjacent and hot-path, so under Section 7 this alias is the LAST to promote: no hot-path alias reaches LIVE until every background alias has completed W5-03 and held steady state for a week.",
|
|
64216
|
+
budget: {
|
|
64217
|
+
basis: "provisional-pre-baseline",
|
|
64218
|
+
monthly_usd: 400,
|
|
64219
|
+
alert_pct: [
|
|
64220
|
+
50,
|
|
64221
|
+
80,
|
|
64222
|
+
100
|
|
64223
|
+
]
|
|
64224
|
+
},
|
|
64225
|
+
routes: [
|
|
64226
|
+
{
|
|
64227
|
+
role: "primary",
|
|
64228
|
+
provider: "groq",
|
|
64229
|
+
model_id: null,
|
|
64230
|
+
model_id_status: "pending-provider-confirmation",
|
|
64231
|
+
model_id_source: null,
|
|
64232
|
+
model_family: "gpt-oss-120B",
|
|
64233
|
+
lumic_model: null,
|
|
64234
|
+
params: {
|
|
64235
|
+
temperature: null,
|
|
64236
|
+
max_output_tokens: null,
|
|
64237
|
+
supports_temperature: true,
|
|
64238
|
+
supports_json_schema: true,
|
|
64239
|
+
supports_tools: true,
|
|
64240
|
+
supports_cache_control: false,
|
|
64241
|
+
supports_vision: false
|
|
64242
|
+
},
|
|
64243
|
+
price_per_mtok: {
|
|
64244
|
+
input: 0.039,
|
|
64245
|
+
output: 0.19,
|
|
64246
|
+
as_of: "2026-09-10",
|
|
64247
|
+
source: "docs/llm-provider-migration.md#3 gpt-oss-120B anchor (DeepInfra price; Groq price unconfirmed pending OI-01)"
|
|
64248
|
+
},
|
|
64249
|
+
notes: "Unreachable until OI-01 is resolved — Groq is not on the W3 onboarding list."
|
|
64250
|
+
},
|
|
64251
|
+
{
|
|
64252
|
+
role: "secondary",
|
|
64253
|
+
provider: "deepinfra",
|
|
64254
|
+
model_id: null,
|
|
64255
|
+
model_id_status: "pending-provider-confirmation",
|
|
64256
|
+
model_id_source: null,
|
|
64257
|
+
model_family: "gpt-oss-120B",
|
|
64258
|
+
lumic_model: null,
|
|
64259
|
+
params: {
|
|
64260
|
+
temperature: null,
|
|
64261
|
+
max_output_tokens: null,
|
|
64262
|
+
supports_temperature: true,
|
|
64263
|
+
supports_json_schema: true,
|
|
64264
|
+
supports_tools: true,
|
|
64265
|
+
supports_cache_control: false,
|
|
64266
|
+
supports_vision: false
|
|
64267
|
+
},
|
|
64268
|
+
price_per_mtok: {
|
|
64269
|
+
input: 0.039,
|
|
64270
|
+
output: 0.19,
|
|
64271
|
+
as_of: "2026-09-10",
|
|
64272
|
+
source: "docs/llm-provider-migration.md#3 DeepInfra anchor; re-verify at W2-02"
|
|
64273
|
+
}
|
|
64274
|
+
},
|
|
64275
|
+
{
|
|
64276
|
+
role: "closed_incumbent",
|
|
64277
|
+
provider: "anthropic",
|
|
64278
|
+
model_id: "claude-haiku-4-5",
|
|
64279
|
+
model_id_status: "confirmed",
|
|
64280
|
+
model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
|
|
64281
|
+
model_family: "Haiku-class",
|
|
64282
|
+
lumic_model: "claude-haiku-4-5",
|
|
64283
|
+
params: {
|
|
64284
|
+
temperature: null,
|
|
64285
|
+
max_output_tokens: 64000,
|
|
64286
|
+
supports_temperature: true,
|
|
64287
|
+
supports_json_schema: true,
|
|
64288
|
+
supports_tools: true,
|
|
64289
|
+
supports_cache_control: true,
|
|
64290
|
+
supports_vision: true,
|
|
64291
|
+
context_window: 200000
|
|
64292
|
+
},
|
|
64293
|
+
price_per_mtok: {
|
|
64294
|
+
input: 1,
|
|
64295
|
+
output: 5,
|
|
64296
|
+
as_of: "2026-09-10",
|
|
64297
|
+
source: "docs/llm-provider-migration.md#3 Haiku-class anchor"
|
|
64298
|
+
}
|
|
64299
|
+
}
|
|
64300
|
+
]
|
|
64301
|
+
},
|
|
64302
|
+
"llm.extract": {
|
|
64303
|
+
workload: "High-volume extraction, classification, log analysis",
|
|
64304
|
+
latency_class: "batch",
|
|
64305
|
+
criticality: "internal",
|
|
64306
|
+
isolation_capable: true,
|
|
64307
|
+
eval_gate: "structured-output",
|
|
64308
|
+
budget: {
|
|
64309
|
+
basis: "provisional-pre-baseline",
|
|
64310
|
+
monthly_usd: 300,
|
|
64311
|
+
alert_pct: [
|
|
64312
|
+
50,
|
|
64313
|
+
80,
|
|
64314
|
+
100
|
|
64315
|
+
]
|
|
64316
|
+
},
|
|
64317
|
+
routes: [
|
|
64318
|
+
{
|
|
64319
|
+
role: "primary",
|
|
64320
|
+
provider: "deepseek",
|
|
64321
|
+
model_id: "deepseek-v4-flash",
|
|
64322
|
+
model_id_status: "confirmed",
|
|
64323
|
+
model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
|
|
64324
|
+
model_family: "DeepSeek V4 Flash",
|
|
64325
|
+
lumic_model: "deepseek-v4-flash",
|
|
64326
|
+
params: {
|
|
64327
|
+
temperature: null,
|
|
64328
|
+
max_output_tokens: 64000,
|
|
64329
|
+
supports_temperature: true,
|
|
64330
|
+
supports_json_schema: true,
|
|
64331
|
+
supports_tools: true,
|
|
64332
|
+
supports_cache_control: true,
|
|
64333
|
+
supports_vision: false,
|
|
64334
|
+
context_window: 1000000
|
|
64335
|
+
},
|
|
64336
|
+
price_per_mtok: {
|
|
64337
|
+
input: 0.14,
|
|
64338
|
+
output: 0.28,
|
|
64339
|
+
as_of: "2026-09-10",
|
|
64340
|
+
source: "docs/llm-provider-migration.md#3 V4 Flash anchor; re-verify at W2-02"
|
|
64341
|
+
}
|
|
64342
|
+
},
|
|
64343
|
+
{
|
|
64344
|
+
role: "secondary",
|
|
64345
|
+
provider: "zai",
|
|
64346
|
+
model_id: null,
|
|
64347
|
+
model_id_status: "pending-provider-confirmation",
|
|
64348
|
+
model_id_source: null,
|
|
64349
|
+
model_family: "GLM 5.3 Flash",
|
|
64350
|
+
lumic_model: null,
|
|
64351
|
+
params: {
|
|
64352
|
+
temperature: null,
|
|
64353
|
+
max_output_tokens: null,
|
|
64354
|
+
supports_temperature: true,
|
|
64355
|
+
supports_json_schema: true,
|
|
64356
|
+
supports_tools: true,
|
|
64357
|
+
supports_cache_control: false,
|
|
64358
|
+
supports_vision: false
|
|
64359
|
+
},
|
|
64360
|
+
price_per_mtok: {
|
|
64361
|
+
input: 1.09,
|
|
64362
|
+
output: 3.43,
|
|
64363
|
+
as_of: "2026-09-10",
|
|
64364
|
+
source: "docs/llm-provider-migration.md#3 GLM first-party anchor; Flash-tier price unconfirmed, re-verify at W2-02"
|
|
64365
|
+
}
|
|
64366
|
+
},
|
|
64367
|
+
{
|
|
64368
|
+
role: "closed_incumbent",
|
|
64369
|
+
provider: "anthropic",
|
|
64370
|
+
model_id: "claude-haiku-4-5",
|
|
64371
|
+
model_id_status: "confirmed",
|
|
64372
|
+
model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
|
|
64373
|
+
model_family: "Haiku-class",
|
|
64374
|
+
lumic_model: "claude-haiku-4-5",
|
|
64375
|
+
params: {
|
|
64376
|
+
temperature: null,
|
|
64377
|
+
max_output_tokens: 64000,
|
|
64378
|
+
supports_temperature: true,
|
|
64379
|
+
supports_json_schema: true,
|
|
64380
|
+
supports_tools: true,
|
|
64381
|
+
supports_cache_control: true,
|
|
64382
|
+
supports_vision: true,
|
|
64383
|
+
context_window: 200000
|
|
64384
|
+
},
|
|
64385
|
+
price_per_mtok: {
|
|
64386
|
+
input: 1,
|
|
64387
|
+
output: 5,
|
|
64388
|
+
as_of: "2026-09-10",
|
|
64389
|
+
source: "docs/llm-provider-migration.md#3 Haiku-class anchor"
|
|
64390
|
+
}
|
|
64391
|
+
}
|
|
64392
|
+
]
|
|
64393
|
+
},
|
|
64394
|
+
"llm.judge": {
|
|
64395
|
+
workload: "Eval judging only",
|
|
64396
|
+
latency_class: "batch",
|
|
64397
|
+
criticality: "internal",
|
|
64398
|
+
isolation_capable: false,
|
|
64399
|
+
pinned: true,
|
|
64400
|
+
eval_gate: "none-pinned-judge",
|
|
64401
|
+
policy_note: "PD-6: pinned and never auto-swapped, and deliberately carries no fallback. A judge that silently failed over would grade one model's output against another model's standard, which would corrupt every gate decided by it — so it fails loud instead.",
|
|
64402
|
+
budget: {
|
|
64403
|
+
basis: "provisional-pre-baseline",
|
|
64404
|
+
monthly_usd: 150,
|
|
64405
|
+
alert_pct: [
|
|
64406
|
+
50,
|
|
64407
|
+
80,
|
|
64408
|
+
100
|
|
64409
|
+
]
|
|
64410
|
+
},
|
|
64411
|
+
routes: [
|
|
64412
|
+
{
|
|
64413
|
+
role: "primary",
|
|
64414
|
+
provider: "anthropic",
|
|
64415
|
+
model_id: "claude-sonnet-4-6",
|
|
64416
|
+
model_id_status: "confirmed",
|
|
64417
|
+
model_id_source: "lumic-utils/src/types/openai-types.ts SUPPORTED_MODELS — already routed on in production",
|
|
64418
|
+
model_family: "Sonnet-class",
|
|
64419
|
+
lumic_model: "claude-sonnet-4-6",
|
|
64420
|
+
params: {
|
|
64421
|
+
temperature: 0,
|
|
64422
|
+
max_output_tokens: 64000,
|
|
64423
|
+
supports_temperature: true,
|
|
64424
|
+
supports_json_schema: true,
|
|
64425
|
+
supports_tools: true,
|
|
64426
|
+
supports_cache_control: true,
|
|
64427
|
+
supports_vision: true,
|
|
64428
|
+
context_window: 1000000
|
|
64429
|
+
},
|
|
64430
|
+
price_per_mtok: {
|
|
64431
|
+
input: 3,
|
|
64432
|
+
output: 15,
|
|
64433
|
+
as_of: "2026-09-10",
|
|
64434
|
+
source: "lumic-utils/src/functions/llm-config.ts anthropicModelCosts"
|
|
64435
|
+
}
|
|
64436
|
+
}
|
|
64437
|
+
]
|
|
64438
|
+
}
|
|
64439
|
+
};
|
|
64440
|
+
var open_items = [
|
|
64441
|
+
{
|
|
64442
|
+
id: "OI-01",
|
|
64443
|
+
question: "Section 3 makes Groq or Cerebras the llm.fast primary, but neither appears in the W3 onboarding provider list (DeepInfra, Fireworks, Z.ai, DeepSeek, aggregator). Which host serves llm.fast's primary leg, and is it added to W3?",
|
|
64444
|
+
blocks: "llm.fast primary reachability; W5-03 canary for llm.fast",
|
|
64445
|
+
owner: "CEO"
|
|
64446
|
+
},
|
|
64447
|
+
{
|
|
64448
|
+
id: "OI-02",
|
|
64449
|
+
question: "Section 3 names a Fable-class primary for llm.agentic, but the lumic model registry's strongest registered Anthropic model is claude-opus-4-7. Register a Fable-class model in @adaptic/lumic-utils, or accept opus-4-7 as the llm.agentic primary?",
|
|
64450
|
+
blocks: "llm.agentic primary fidelity to Section 3",
|
|
64451
|
+
owner: "CEO"
|
|
64452
|
+
},
|
|
64453
|
+
{
|
|
64454
|
+
id: "OI-03",
|
|
64455
|
+
question: "Every alias budget is provisional-pre-baseline because Section 9's baseline monthly spend is still FILL. Which caps apply once W1-03 produces the billing baseline?",
|
|
64456
|
+
blocks: "PD-7 layer 3 calibration; Section 9 savings target measurement",
|
|
64457
|
+
owner: "CEO"
|
|
64458
|
+
},
|
|
64459
|
+
{
|
|
64460
|
+
id: "OI-04",
|
|
64461
|
+
question: "llm.agentic's serving chain is one leg long because Section 3 makes its primary the closed incumbent and its only secondary shadow-eval-only. An Anthropic outage therefore leaves the alias with no configured fallback. Widen it (a second closed provider, or promoting the Kimi leg out of shadow), or accept the single leg?",
|
|
64462
|
+
blocks: "llm.agentic availability under a primary-provider outage",
|
|
64463
|
+
owner: "CEO"
|
|
64464
|
+
}
|
|
64465
|
+
];
|
|
64466
|
+
var rawTable = {
|
|
64467
|
+
schema_version: schema_version,
|
|
64468
|
+
policy_source: policy_source,
|
|
64469
|
+
revised: revised,
|
|
64470
|
+
defaults: defaults,
|
|
64471
|
+
providers: providers,
|
|
64472
|
+
aliases: aliases,
|
|
64473
|
+
open_items: open_items
|
|
64474
|
+
};
|
|
64475
|
+
|
|
64476
|
+
/**
|
|
64477
|
+
* Typed access to the canonical alias route table.
|
|
64478
|
+
*
|
|
64479
|
+
* The table is imported rather than fetched so an alias always resolves, even
|
|
64480
|
+
* when the gateway is unreachable. A client that could only learn its routes
|
|
64481
|
+
* from the gateway would have no way to fall back when the gateway itself is
|
|
64482
|
+
* the thing that failed, which would make the mandatory fallback chain of PD-3
|
|
64483
|
+
* conditional on the very component it exists to survive.
|
|
64484
|
+
*
|
|
64485
|
+
* Resolution is deliberately conservative in three ways. An unknown alias is an
|
|
64486
|
+
* error rather than a default, because a typo silently served by whichever
|
|
64487
|
+
* model happened to be configured is worse than a loud failure. A route whose
|
|
64488
|
+
* vendor model id is unconfirmed is excluded, because a guessed id fails at the
|
|
64489
|
+
* first live call rather than at review. And a shadow-only route is never
|
|
64490
|
+
* served, because a candidate that can answer a live request has stopped being
|
|
64491
|
+
* a candidate.
|
|
64492
|
+
*
|
|
64493
|
+
* @module llm/route-table
|
|
64494
|
+
*/
|
|
64495
|
+
/** Chain order, fixed by role rather than by authoring order in the file. */
|
|
64496
|
+
const ROLE_ORDER = ["primary", "secondary", "closed_incumbent"];
|
|
64497
|
+
/** Suffix naming an alias's research-isolated variant (PD-9). */
|
|
64498
|
+
const ISOLATED_SUFFIX = ".isolated";
|
|
64499
|
+
/**
|
|
64500
|
+
* The canonical route table.
|
|
64501
|
+
*
|
|
64502
|
+
* Exposed as a readonly view so a consumer can inspect routing without being
|
|
64503
|
+
* able to mutate it. A table that could be edited at runtime would let one
|
|
64504
|
+
* caller change every other caller's routing.
|
|
64505
|
+
*/
|
|
64506
|
+
const routeTable = rawTable;
|
|
64507
|
+
/**
|
|
64508
|
+
* Every alias the table defines.
|
|
64509
|
+
*
|
|
64510
|
+
* @returns The alias names, sorted for stable iteration.
|
|
64511
|
+
*/
|
|
64512
|
+
function listAliases() {
|
|
64513
|
+
return Object.keys(routeTable.aliases).sort();
|
|
64514
|
+
}
|
|
64515
|
+
/**
|
|
64516
|
+
* Look up an alias definition.
|
|
64517
|
+
*
|
|
64518
|
+
* @param alias The alias to resolve.
|
|
64519
|
+
* @returns Its definition.
|
|
64520
|
+
* @throws When the alias is not defined by the route table.
|
|
64521
|
+
*/
|
|
64522
|
+
function aliasDefinition(alias) {
|
|
64523
|
+
const definition = routeTable.aliases[alias];
|
|
64524
|
+
if (definition === undefined) {
|
|
64525
|
+
throw new UnknownAliasError(alias, listAliases());
|
|
64526
|
+
}
|
|
64527
|
+
return definition;
|
|
64528
|
+
}
|
|
64529
|
+
/**
|
|
64530
|
+
* Look up a provider registry entry.
|
|
64531
|
+
*
|
|
64532
|
+
* @param name The provider key.
|
|
64533
|
+
* @returns Its registry entry.
|
|
64534
|
+
* @throws When the provider is not registered.
|
|
64535
|
+
*/
|
|
64536
|
+
function providerEntry(name) {
|
|
64537
|
+
const provider = routeTable.providers[name];
|
|
64538
|
+
if (provider === undefined) {
|
|
64539
|
+
throw new Error(`route table names provider "${name}", which is not in the provider registry`);
|
|
64540
|
+
}
|
|
64541
|
+
return provider;
|
|
64542
|
+
}
|
|
64543
|
+
/** Thrown when a caller names an alias the route table does not define. */
|
|
64544
|
+
class UnknownAliasError extends Error {
|
|
64545
|
+
/** The alias that was requested. */
|
|
64546
|
+
alias;
|
|
64547
|
+
/** The aliases that do exist, so the message is actionable. */
|
|
64548
|
+
known;
|
|
64549
|
+
/**
|
|
64550
|
+
* @param alias The unrecognised alias.
|
|
64551
|
+
* @param known The aliases the table defines.
|
|
64552
|
+
*/
|
|
64553
|
+
constructor(alias, known) {
|
|
64554
|
+
super(`unknown LLM alias "${alias}". Known aliases: ${known.join(", ")}. ` +
|
|
64555
|
+
"Application code names aliases only; adding one is a change to the route table.");
|
|
64556
|
+
this.name = "UnknownAliasError";
|
|
64557
|
+
this.alias = alias;
|
|
64558
|
+
this.known = known;
|
|
64559
|
+
}
|
|
64560
|
+
}
|
|
64561
|
+
/** Thrown when an alias exists but has no leg that can currently serve a caller. */
|
|
64562
|
+
class NoServableRouteError extends Error {
|
|
64563
|
+
/** The alias that could not be served. */
|
|
64564
|
+
alias;
|
|
64565
|
+
/** Why each of its legs was excluded, in chain order. */
|
|
64566
|
+
exclusions;
|
|
64567
|
+
/**
|
|
64568
|
+
* @param alias The alias.
|
|
64569
|
+
* @param exclusions Per-leg reasons, in chain order.
|
|
64570
|
+
*/
|
|
64571
|
+
constructor(alias, exclusions) {
|
|
64572
|
+
super(`alias "${alias}" has no servable route. Legs excluded: ${exclusions.join("; ")}`);
|
|
64573
|
+
this.name = "NoServableRouteError";
|
|
64574
|
+
this.alias = alias;
|
|
64575
|
+
this.exclusions = exclusions;
|
|
64576
|
+
}
|
|
64577
|
+
}
|
|
64578
|
+
/**
|
|
64579
|
+
* Order an alias's routes into the chain the client walks.
|
|
64580
|
+
*
|
|
64581
|
+
* @param definition The alias definition.
|
|
64582
|
+
* @returns Routes ordered primary -> secondary -> closed incumbent.
|
|
64583
|
+
*/
|
|
64584
|
+
function orderedRoutes(definition) {
|
|
64585
|
+
return [...definition.routes].sort((a, b) => ROLE_ORDER.indexOf(a.role) - ROLE_ORDER.indexOf(b.role));
|
|
64586
|
+
}
|
|
64587
|
+
/**
|
|
64588
|
+
* Stable identity for one leg, used to key its circuit breaker and its metrics.
|
|
64589
|
+
*
|
|
64590
|
+
* Keyed by alias, isolation and role rather than by provider and model, because
|
|
64591
|
+
* the breaker guards a position in a chain: reverting an alias to a different
|
|
64592
|
+
* model at the same position should inherit that position's health rather than
|
|
64593
|
+
* start blind, and the isolated variant must never share a breaker with the
|
|
64594
|
+
* shared one.
|
|
64595
|
+
*
|
|
64596
|
+
* @param alias The alias.
|
|
64597
|
+
* @param isolated Whether this is the isolated variant.
|
|
64598
|
+
* @param role The leg's role.
|
|
64599
|
+
* @returns The route key.
|
|
64600
|
+
*/
|
|
64601
|
+
function routeKeyFor(alias, isolated, role) {
|
|
64602
|
+
return `${alias}${isolated ? ISOLATED_SUFFIX : ""}#${role}`;
|
|
64603
|
+
}
|
|
64604
|
+
/**
|
|
64605
|
+
* Resolve an alias to the ordered chain of legs that can serve it today.
|
|
64606
|
+
*
|
|
64607
|
+
* Exclusions are returned rather than discarded so an exhausted chain can say
|
|
64608
|
+
* why each leg was unavailable. "No route available" without that detail sends
|
|
64609
|
+
* an operator to read config; with it, the answer is in the error.
|
|
64610
|
+
*
|
|
64611
|
+
* @param alias The alias to resolve.
|
|
64612
|
+
* @param options Resolution options.
|
|
64613
|
+
* @param options.isolated Route through the isolated variant (PD-9).
|
|
64614
|
+
* @param options.timeoutMsOverride Per-leg budget override, never wider than the caller's own deadline.
|
|
64615
|
+
* @returns The resolved chain.
|
|
64616
|
+
* @throws {UnknownAliasError} When the alias is not defined.
|
|
64617
|
+
*/
|
|
64618
|
+
function resolveChain(alias, options = {}) {
|
|
64619
|
+
const definition = aliasDefinition(alias);
|
|
64620
|
+
const isolated = options.isolated === true;
|
|
64621
|
+
if (isolated && !definition.isolation_capable) {
|
|
64622
|
+
throw new Error(`alias "${alias}" is not isolation-capable, so it has no isolated variant. ` +
|
|
64623
|
+
"PD-9 forbids serving isolated work from a shared route, so this fails rather than falling back.");
|
|
64624
|
+
}
|
|
64625
|
+
const timeoutMs = options.timeoutMsOverride ??
|
|
64626
|
+
routeTable.defaults.request_timeout_ms[definition.latency_class];
|
|
64627
|
+
const resolved = [];
|
|
64628
|
+
const exclusions = [];
|
|
64629
|
+
for (const route of orderedRoutes(definition)) {
|
|
64630
|
+
const provider = providerEntry(route.provider);
|
|
64631
|
+
if (route.shadow_only === true) {
|
|
64632
|
+
exclusions.push({
|
|
64633
|
+
role: route.role,
|
|
64634
|
+
provider: route.provider,
|
|
64635
|
+
reason: "shadow-only: configured and scored, never served to a caller",
|
|
64636
|
+
});
|
|
64637
|
+
continue;
|
|
64638
|
+
}
|
|
64639
|
+
if (route.model_id_status !== "confirmed" || route.model_id === null) {
|
|
64640
|
+
exclusions.push({
|
|
64641
|
+
role: route.role,
|
|
64642
|
+
provider: route.provider,
|
|
64643
|
+
reason: `model id unconfirmed (${route.model_family}); it is transcribed from the provider console at onboarding, never guessed`,
|
|
64644
|
+
});
|
|
64645
|
+
continue;
|
|
64646
|
+
}
|
|
64647
|
+
if (provider.account_status !== "live") {
|
|
64648
|
+
exclusions.push({
|
|
64649
|
+
role: route.role,
|
|
64650
|
+
provider: route.provider,
|
|
64651
|
+
reason: `provider account status is "${provider.account_status}"`,
|
|
64652
|
+
});
|
|
64653
|
+
continue;
|
|
64654
|
+
}
|
|
64655
|
+
resolved.push({
|
|
64656
|
+
alias,
|
|
64657
|
+
isolated,
|
|
64658
|
+
role: route.role,
|
|
64659
|
+
providerName: route.provider,
|
|
64660
|
+
provider,
|
|
64661
|
+
modelId: route.model_id,
|
|
64662
|
+
lumicModel: route.lumic_model ?? null,
|
|
64663
|
+
params: route.params ?? {},
|
|
64664
|
+
routeKey: routeKeyFor(alias, isolated, route.role),
|
|
64665
|
+
timeoutMs,
|
|
64666
|
+
retriesPerLeg: routeTable.defaults.retries_per_leg,
|
|
64667
|
+
});
|
|
64668
|
+
}
|
|
64669
|
+
return { alias, isolated, routes: resolved, exclusions };
|
|
64670
|
+
}
|
|
64671
|
+
/**
|
|
64672
|
+
* The permanent closed-incumbent leg of an alias (PD-11).
|
|
64673
|
+
*
|
|
64674
|
+
* Found by provider tier rather than by role, because an alias whose primary is
|
|
64675
|
+
* already a closed vendor has that primary as its revert target. Both shapes
|
|
64676
|
+
* therefore hold the same guarantee: there is always a configured route that
|
|
64677
|
+
* needs no new account and no code change to fall back to.
|
|
64678
|
+
*
|
|
64679
|
+
* @param chain A resolved chain.
|
|
64680
|
+
* @returns The closed leg, or undefined when none is currently servable.
|
|
64681
|
+
*/
|
|
64682
|
+
function closedIncumbentLeg(chain) {
|
|
64683
|
+
return chain.routes.find((route) => route.provider.tier === "closed");
|
|
64684
|
+
}
|
|
64685
|
+
/**
|
|
64686
|
+
* The name the gateway knows a leg by.
|
|
64687
|
+
*
|
|
64688
|
+
* The head of a chain is addressed by the bare alias so a caller never learns a
|
|
64689
|
+
* leg name; every other leg carries the derived name the renderer gives it.
|
|
64690
|
+
* Keeping this derivation beside the resolver — rather than in the transport —
|
|
64691
|
+
* is what stops the client and the gateway config from disagreeing about a name.
|
|
64692
|
+
*
|
|
64693
|
+
* @param route The resolved leg.
|
|
64694
|
+
* @param chain The chain it belongs to.
|
|
64695
|
+
* @returns The gateway `model_name`.
|
|
64696
|
+
*/
|
|
64697
|
+
function gatewayModelNameFor(route, chain) {
|
|
64698
|
+
const base = `${route.alias}${route.isolated ? ISOLATED_SUFFIX : ""}`;
|
|
64699
|
+
return chain.routes[0]?.routeKey === route.routeKey
|
|
64700
|
+
? base
|
|
64701
|
+
: `${base}.fallback.${route.role}`;
|
|
64702
|
+
}
|
|
64703
|
+
|
|
64704
|
+
/**
|
|
64705
|
+
* Bounded validate-and-retry for structured output.
|
|
64706
|
+
*
|
|
64707
|
+
* A model asked for a schema-shaped answer sometimes returns something close
|
|
64708
|
+
* but wrong. Feeding the validator's own complaint back once fixes most of
|
|
64709
|
+
* those, because the model can see precisely what it got wrong. A second retry
|
|
64710
|
+
* almost never helps: if the model is still wrong after being told exactly what
|
|
64711
|
+
* was wrong, it is persistently wrong for this prompt, and further attempts
|
|
64712
|
+
* spend budget and latency to arrive at the same place.
|
|
64713
|
+
*
|
|
64714
|
+
* The retry lives ahead of the fallback chain rather than inside it. A payload
|
|
64715
|
+
* that fails validation is evidence about the PROMPT, not about the provider's
|
|
64716
|
+
* health, so it must not open a circuit breaker or advance to the next leg —
|
|
64717
|
+
* the next leg would receive the same prompt and be no likelier to satisfy it.
|
|
64718
|
+
*
|
|
64719
|
+
* Exhaustion throws. Returning a partial or defaulted object would hand the
|
|
64720
|
+
* caller a well-typed value that no model ever produced, and the resulting
|
|
64721
|
+
* decision would be made on invented data with nothing to show it.
|
|
64722
|
+
*
|
|
64723
|
+
* @module llm/schema-retry
|
|
64724
|
+
*/
|
|
64725
|
+
/** Attempts allowed: the first, plus exactly one feedback retry. */
|
|
64726
|
+
const MAX_ATTEMPTS = 2;
|
|
64727
|
+
/** Characters of an invalid payload echoed back to the model. */
|
|
64728
|
+
const PAYLOAD_EXCERPT = 2000;
|
|
64729
|
+
/**
|
|
64730
|
+
* Thrown when both attempts failed validation.
|
|
64731
|
+
*
|
|
64732
|
+
* Carries both rejection reasons, because the pair is what distinguishes a
|
|
64733
|
+
* flaky answer from a prompt the model cannot satisfy: two different complaints
|
|
64734
|
+
* suggest instability, while the same complaint twice points at the prompt or
|
|
64735
|
+
* the schema.
|
|
64736
|
+
*/
|
|
64737
|
+
class SchemaRetryExhaustedError extends Error {
|
|
64738
|
+
/** Why the first attempt was rejected. */
|
|
64739
|
+
firstReason;
|
|
64740
|
+
/** Why the retry was rejected. */
|
|
64741
|
+
secondReason;
|
|
64742
|
+
/** Usage spent across both attempts, so the spend is still accounted for. */
|
|
64743
|
+
totalUsage;
|
|
64744
|
+
/**
|
|
64745
|
+
* @param firstReason Validator's complaint about attempt one.
|
|
64746
|
+
* @param secondReason Validator's complaint about attempt two.
|
|
64747
|
+
* @param totalUsage Usage across both attempts.
|
|
64748
|
+
*/
|
|
64749
|
+
constructor(firstReason, secondReason, totalUsage) {
|
|
64750
|
+
super("LLM structured output failed validation twice. " +
|
|
64751
|
+
`First: ${firstReason}. After feedback: ${secondReason}. ` +
|
|
64752
|
+
"No value is returned: a defaulted object would be data no model produced.");
|
|
64753
|
+
this.name = "SchemaRetryExhaustedError";
|
|
64754
|
+
this.firstReason = firstReason;
|
|
64755
|
+
this.secondReason = secondReason;
|
|
64756
|
+
this.totalUsage = totalUsage;
|
|
64757
|
+
}
|
|
64758
|
+
}
|
|
64759
|
+
/**
|
|
64760
|
+
* Build the retry prompt.
|
|
64761
|
+
*
|
|
64762
|
+
* The rejected payload is echoed back alongside the complaint, because a model
|
|
64763
|
+
* asked to "fix the error" without seeing what it produced will usually
|
|
64764
|
+
* regenerate from scratch and reproduce the same mistake.
|
|
64765
|
+
*
|
|
64766
|
+
* @param originalPrompt The prompt that produced the invalid payload.
|
|
64767
|
+
* @param rejected The payload that failed.
|
|
64768
|
+
* @param reason The validator's complaint.
|
|
64769
|
+
* @returns The retry prompt.
|
|
64770
|
+
*/
|
|
64771
|
+
function buildRetryPrompt(originalPrompt, rejected, reason) {
|
|
64772
|
+
const excerpt = typeof rejected === "string"
|
|
64773
|
+
? rejected
|
|
64774
|
+
: JSON.stringify(rejected, null, 2) ?? String(rejected);
|
|
64775
|
+
return [
|
|
64776
|
+
originalPrompt,
|
|
64777
|
+
"",
|
|
64778
|
+
"Your previous response was rejected by a schema validator.",
|
|
64779
|
+
"",
|
|
64780
|
+
"Previous response:",
|
|
64781
|
+
excerpt.slice(0, PAYLOAD_EXCERPT),
|
|
64782
|
+
"",
|
|
64783
|
+
`Validator rejection: ${reason}`,
|
|
64784
|
+
"",
|
|
64785
|
+
"Return a corrected response that satisfies the schema. Return only the corrected response.",
|
|
64786
|
+
].join("\n");
|
|
64787
|
+
}
|
|
64788
|
+
/**
|
|
64789
|
+
* Run a call with one validator-feedback retry.
|
|
64790
|
+
*
|
|
64791
|
+
* @param options Retry options.
|
|
64792
|
+
* @param options.prompt The original prompt.
|
|
64793
|
+
* @param options.validate Validator applied to each attempt's payload.
|
|
64794
|
+
* @param options.call Executes one attempt with the given prompt.
|
|
64795
|
+
* @returns The validated value with combined usage.
|
|
64796
|
+
* @throws {SchemaRetryExhaustedError} When both attempts fail validation.
|
|
64797
|
+
*/
|
|
64798
|
+
async function callWithValidation(options) {
|
|
64799
|
+
const first = await options.call(options.prompt);
|
|
64800
|
+
const firstOutcome = options.validate(first.response);
|
|
64801
|
+
if (firstOutcome.ok) {
|
|
64802
|
+
return {
|
|
64803
|
+
value: firstOutcome.value,
|
|
64804
|
+
attempts: 1,
|
|
64805
|
+
response: first,
|
|
64806
|
+
totalUsage: first.usage,
|
|
64807
|
+
};
|
|
64808
|
+
}
|
|
64809
|
+
const retryPrompt = buildRetryPrompt(options.prompt, first.response, firstOutcome.reason);
|
|
64810
|
+
const second = await options.call(retryPrompt);
|
|
64811
|
+
const combined = sumUsage(first.usage, second.usage);
|
|
64812
|
+
const secondOutcome = options.validate(second.response);
|
|
64813
|
+
if (secondOutcome.ok) {
|
|
64814
|
+
return {
|
|
64815
|
+
value: secondOutcome.value,
|
|
64816
|
+
attempts: MAX_ATTEMPTS,
|
|
64817
|
+
response: second,
|
|
64818
|
+
totalUsage: combined,
|
|
64819
|
+
};
|
|
64820
|
+
}
|
|
64821
|
+
throw new SchemaRetryExhaustedError(firstOutcome.reason, secondOutcome.reason, combined);
|
|
64822
|
+
}
|
|
64823
|
+
|
|
64824
|
+
/**
|
|
64825
|
+
* Degraded direct transport: the path used when the gateway itself is gone.
|
|
64826
|
+
*
|
|
64827
|
+
* Routing every call through one proxy concentrates a great deal of value — one
|
|
64828
|
+
* place to swap a model, one place to bound spend, one place to see cost. It
|
|
64829
|
+
* also concentrates risk: without this path, a gateway outage would take every
|
|
64830
|
+
* LLM call in the system down at once, which is a worse failure than any of the
|
|
64831
|
+
* provider outages the gateway exists to survive.
|
|
64832
|
+
*
|
|
64833
|
+
* Two constraints keep this a safety net rather than a second routing policy.
|
|
64834
|
+
* It serves CLOSED-tier legs only, so the degraded path can never be the thing
|
|
64835
|
+
* that silently promotes an open-weight model past its evaluation gates. And it
|
|
64836
|
+
* resolves the model from the same route table the gateway is rendered from, so
|
|
64837
|
+
* degraded traffic reaches the same model the gateway would have chosen.
|
|
64838
|
+
*
|
|
64839
|
+
* The provider SDK is reached through an injected caller, resolved lazily. A
|
|
64840
|
+
* static import would make `@adaptic/utils` load a provider SDK for every
|
|
64841
|
+
* consumer, including those that never make an LLM call, and would harden a
|
|
64842
|
+
* package cycle that is currently only a declaration.
|
|
64843
|
+
*
|
|
64844
|
+
* @module llm/transports/direct
|
|
64845
|
+
*/
|
|
64846
|
+
/**
|
|
64847
|
+
* Thrown when the degraded path is asked to serve a leg it must not serve.
|
|
64848
|
+
*
|
|
64849
|
+
* Refusing loudly rather than serving the leg anyway is the point: the whole
|
|
64850
|
+
* value of restricting this path is lost if it quietly widens under pressure,
|
|
64851
|
+
* and pressure is exactly when it runs.
|
|
64852
|
+
*/
|
|
64853
|
+
class DirectTransportRefusedError extends Error {
|
|
64854
|
+
/**
|
|
64855
|
+
* @param routeKey The leg that was refused.
|
|
64856
|
+
* @param reason Why it cannot be served directly.
|
|
64857
|
+
*/
|
|
64858
|
+
constructor(routeKey, reason) {
|
|
64859
|
+
super(`the degraded direct transport refuses route ${routeKey}: ${reason}. ` +
|
|
64860
|
+
"It exists so a gateway outage does not stop every LLM call, not as a second routing policy.");
|
|
64861
|
+
this.name = "DirectTransportRefusedError";
|
|
64862
|
+
}
|
|
64863
|
+
}
|
|
64864
|
+
/**
|
|
64865
|
+
* Build the degraded direct transport.
|
|
64866
|
+
*
|
|
64867
|
+
* @param config Transport configuration.
|
|
64868
|
+
* @returns A transport that reaches a closed-tier provider without the gateway.
|
|
64869
|
+
*/
|
|
64870
|
+
function createDirectTransport(config) {
|
|
64871
|
+
return {
|
|
64872
|
+
name: "direct",
|
|
64873
|
+
async execute(request) {
|
|
64874
|
+
const { route } = request;
|
|
64875
|
+
if (route.provider.tier !== "closed") {
|
|
64876
|
+
throw new DirectTransportRefusedError(route.routeKey, `provider "${route.providerName}" is ${route.provider.tier}-tier; only a closed incumbent may be served without the gateway, so a degraded call can never promote an open route past its evaluation gates`);
|
|
64877
|
+
}
|
|
64878
|
+
if (route.lumicModel === null) {
|
|
64879
|
+
throw new DirectTransportRefusedError(route.routeKey, "the route names no lumic_model, so there is no registered model to call directly");
|
|
64880
|
+
}
|
|
64881
|
+
const call = await config.resolveCaller();
|
|
64882
|
+
const result = await call(request.content, request.responseFormat, {
|
|
64883
|
+
...request.params,
|
|
64884
|
+
model: route.lumicModel,
|
|
64885
|
+
signal: request.signal,
|
|
64886
|
+
timeout: route.timeoutMs,
|
|
64887
|
+
});
|
|
64888
|
+
return {
|
|
64889
|
+
response: result.response,
|
|
64890
|
+
usage: readUsage$1(result.usage, request),
|
|
64891
|
+
tool_calls: Array.isArray(result.tool_calls)
|
|
64892
|
+
? result.tool_calls
|
|
64893
|
+
: undefined,
|
|
64894
|
+
};
|
|
64895
|
+
},
|
|
64896
|
+
};
|
|
64897
|
+
}
|
|
64898
|
+
/**
|
|
64899
|
+
* Normalise the incumbent client's usage shape.
|
|
64900
|
+
*
|
|
64901
|
+
* Missing counts stay zero rather than being estimated, for the same reason
|
|
64902
|
+
* they do on the gateway path: an invented token count flows straight into the
|
|
64903
|
+
* budget accounting the spend controls depend on.
|
|
64904
|
+
*
|
|
64905
|
+
* @param usage The incumbent client's usage, if any.
|
|
64906
|
+
* @param request The request it answers.
|
|
64907
|
+
* @returns The normalised usage record.
|
|
64908
|
+
*/
|
|
64909
|
+
function readUsage$1(usage, request) {
|
|
64910
|
+
return {
|
|
64911
|
+
prompt_tokens: usage?.prompt_tokens ?? 0,
|
|
64912
|
+
completion_tokens: usage?.completion_tokens ?? 0,
|
|
64913
|
+
reasoning_tokens: usage?.reasoning_tokens,
|
|
64914
|
+
cached_tokens: usage?.cached_tokens,
|
|
64915
|
+
provider: usage?.provider ?? request.route.providerName,
|
|
64916
|
+
model: usage?.model ?? request.route.modelId,
|
|
64917
|
+
cost: usage?.cost ?? 0,
|
|
64918
|
+
};
|
|
64919
|
+
}
|
|
64920
|
+
/**
|
|
64921
|
+
* Package providing the default provider client, resolved at runtime.
|
|
64922
|
+
*
|
|
64923
|
+
* Assembled rather than written as a literal so the module specifier is opaque
|
|
64924
|
+
* to the compiler and the bundler. That is not a trick to dodge a type error:
|
|
64925
|
+
* this package is genuinely OPTIONAL. The stable lineage of `@adaptic/utils`
|
|
64926
|
+
* does not depend on `@adaptic/lumic-utils` at all, the transport is injectable
|
|
64927
|
+
* precisely so a consumer can supply its own, and the default exists only as a
|
|
64928
|
+
* convenience for consumers that already have it installed. A static specifier
|
|
64929
|
+
* would assert a dependency that does not exist and would harden the
|
|
64930
|
+
* utils/lumic-utils package cycle from a declaration into a build-time fact.
|
|
64931
|
+
*/
|
|
64932
|
+
const DEFAULT_PROVIDER_CLIENT_PACKAGE = ["@adaptic", "lumic-utils"].join("/");
|
|
64933
|
+
/**
|
|
64934
|
+
* The default caller: the incumbent client in `@adaptic/lumic-utils`.
|
|
64935
|
+
*
|
|
64936
|
+
* Resolved with a dynamic import so this module has no load-time dependency on
|
|
64937
|
+
* that package. When it cannot be loaded the failure names the degraded path
|
|
64938
|
+
* explicitly, because "cannot find module" during an outage is otherwise a
|
|
64939
|
+
* confusing second mystery on top of the first — and a consumer that does not
|
|
64940
|
+
* ship that package is expected to register its own transport rather than to
|
|
64941
|
+
* discover this at the moment the gateway fails.
|
|
64942
|
+
*
|
|
64943
|
+
* @returns The provider-calling function.
|
|
64944
|
+
*/
|
|
64945
|
+
async function resolveDefaultDirectCaller() {
|
|
64946
|
+
try {
|
|
64947
|
+
const lumic = (await import(
|
|
64948
|
+
/* @vite-ignore */ DEFAULT_PROVIDER_CLIENT_PACKAGE));
|
|
64949
|
+
const call = lumic.lumic?.llm?.call;
|
|
64950
|
+
if (typeof call !== "function") {
|
|
64951
|
+
throw new Error(`${DEFAULT_PROVIDER_CLIENT_PACKAGE} exposes no lumic.llm.call`);
|
|
64952
|
+
}
|
|
64953
|
+
return call;
|
|
64954
|
+
}
|
|
64955
|
+
catch (error) {
|
|
64956
|
+
throw new Error("the degraded direct transport could not load its provider client: " +
|
|
64957
|
+
`${error instanceof Error ? error.message : String(error)}. ` +
|
|
64958
|
+
"Register a direct transport explicitly via configureLlmClient() where the default is unavailable.");
|
|
64959
|
+
}
|
|
64960
|
+
}
|
|
64961
|
+
|
|
64962
|
+
/**
|
|
64963
|
+
* Gateway transport: the normal path for every LLM call.
|
|
64964
|
+
*
|
|
64965
|
+
* The client sends an alias to the LiteLLM proxy and the proxy resolves it. The
|
|
64966
|
+
* vendor model string therefore exists only in the gateway's configuration and
|
|
64967
|
+
* never in application code (PD-5), which is what makes a model swap a config
|
|
64968
|
+
* change rather than a deploy.
|
|
64969
|
+
*
|
|
64970
|
+
* The client still walks its own chain on top of the gateway's, and the
|
|
64971
|
+
* duplication is deliberate. The gateway's fallbacks cover a provider being
|
|
64972
|
+
* down; the client's cover the gateway being down. Only one of those two can
|
|
64973
|
+
* cover the other, so the outer chain is the one that must exist.
|
|
64974
|
+
*
|
|
64975
|
+
* The gateway key is read from the environment by NAME at call time and never
|
|
64976
|
+
* stored, logged, or included in an error (PD-2). Reading it per call rather
|
|
64977
|
+
* than caching it at import means a rotation takes effect without a restart.
|
|
64978
|
+
*
|
|
64979
|
+
* @module llm/transports/gateway
|
|
64980
|
+
*/
|
|
64981
|
+
/** HTTP statuses that mean "try the next leg" rather than "this request is wrong". */
|
|
64982
|
+
const RETRYABLE_STATUSES = new Set([408, 409, 425, 429, 500, 502, 503, 504]);
|
|
64983
|
+
/** Maximum characters of an error body echoed into a message. */
|
|
64984
|
+
const ERROR_BODY_EXCERPT = 400;
|
|
64985
|
+
/**
|
|
64986
|
+
* Thrown when the gateway itself is unreachable, as opposed to a provider
|
|
64987
|
+
* behind it failing.
|
|
64988
|
+
*
|
|
64989
|
+
* The distinction is what licenses the degraded direct path: a 502 from a
|
|
64990
|
+
* provider means try the next leg, while a connection refused from the proxy
|
|
64991
|
+
* means the whole gateway is gone and the chain cannot be walked through it at
|
|
64992
|
+
* all.
|
|
64993
|
+
*/
|
|
64994
|
+
class GatewayUnreachableError extends Error {
|
|
64995
|
+
/**
|
|
64996
|
+
* @param baseUrl The gateway that could not be reached.
|
|
64997
|
+
* @param cause The underlying transport error.
|
|
64998
|
+
*/
|
|
64999
|
+
constructor(baseUrl, cause) {
|
|
65000
|
+
super(`LLM gateway at ${baseUrl} is unreachable: ${cause instanceof Error ? cause.message : String(cause)}`);
|
|
65001
|
+
this.name = "GatewayUnreachableError";
|
|
65002
|
+
}
|
|
65003
|
+
}
|
|
65004
|
+
/** Thrown when the gateway answered, but with a failure. */
|
|
65005
|
+
class GatewayResponseError extends Error {
|
|
65006
|
+
/** The HTTP status. */
|
|
65007
|
+
status;
|
|
65008
|
+
/** Whether advancing to the next leg could plausibly help. */
|
|
65009
|
+
retryable;
|
|
65010
|
+
/**
|
|
65011
|
+
* @param status The HTTP status.
|
|
65012
|
+
* @param body A bounded excerpt of the response body.
|
|
65013
|
+
*/
|
|
65014
|
+
constructor(status, body) {
|
|
65015
|
+
super(`LLM gateway returned ${status}: ${body.slice(0, ERROR_BODY_EXCERPT)}`);
|
|
65016
|
+
this.name = "GatewayResponseError";
|
|
65017
|
+
this.status = status;
|
|
65018
|
+
this.retryable = RETRYABLE_STATUSES.has(status);
|
|
65019
|
+
}
|
|
65020
|
+
}
|
|
65021
|
+
/**
|
|
65022
|
+
* Read the gateway key from the environment by name.
|
|
65023
|
+
*
|
|
65024
|
+
* @param envVar The variable's NAME.
|
|
65025
|
+
* @returns The key.
|
|
65026
|
+
* @throws When the variable is unset, because an unauthenticated call would
|
|
65027
|
+
* reach the gateway as an anonymous caller and be rejected there anyway —
|
|
65028
|
+
* later, and with a less useful message.
|
|
65029
|
+
*/
|
|
65030
|
+
function readGatewayKey(envVar) {
|
|
65031
|
+
const value = process.env[envVar];
|
|
65032
|
+
if (value === undefined || value.length === 0) {
|
|
65033
|
+
throw new Error(`${envVar} is unset, so the LLM gateway cannot be authenticated against. ` +
|
|
65034
|
+
"Provision it from the secrets manager; it is never read from a file or a default.");
|
|
65035
|
+
}
|
|
65036
|
+
return value;
|
|
65037
|
+
}
|
|
65038
|
+
/**
|
|
65039
|
+
* Extract usage from a chat-completion response.
|
|
65040
|
+
*
|
|
65041
|
+
* Absent counts stay zero rather than being estimated. A fabricated token count
|
|
65042
|
+
* would flow straight into the budget accounting that the spend controls are
|
|
65043
|
+
* built on, and a budget computed from invented numbers is worse than one that
|
|
65044
|
+
* knows it is missing a call.
|
|
65045
|
+
*
|
|
65046
|
+
* @param payload The parsed response body.
|
|
65047
|
+
* @param request The request it answers.
|
|
65048
|
+
* @returns The usage record.
|
|
65049
|
+
*/
|
|
65050
|
+
function readUsage(payload, request) {
|
|
65051
|
+
const usage = (payload.usage ?? {});
|
|
65052
|
+
const details = (usage.prompt_tokens_details ?? {});
|
|
65053
|
+
const cached = details.cached_tokens;
|
|
65054
|
+
const reasoningDetails = (usage.completion_tokens_details ?? {});
|
|
65055
|
+
const reasoning = reasoningDetails.reasoning_tokens;
|
|
65056
|
+
return {
|
|
65057
|
+
prompt_tokens: typeof usage.prompt_tokens === "number" ? usage.prompt_tokens : 0,
|
|
65058
|
+
completion_tokens: typeof usage.completion_tokens === "number" ? usage.completion_tokens : 0,
|
|
65059
|
+
reasoning_tokens: typeof reasoning === "number" ? reasoning : undefined,
|
|
65060
|
+
cached_tokens: typeof cached === "number" ? cached : undefined,
|
|
65061
|
+
provider: request.route.providerName,
|
|
65062
|
+
model: request.route.modelId,
|
|
65063
|
+
cost: typeof usage.response_cost === "number" ? usage.response_cost : 0,
|
|
65064
|
+
};
|
|
65065
|
+
}
|
|
65066
|
+
/**
|
|
65067
|
+
* Build the gateway transport.
|
|
65068
|
+
*
|
|
65069
|
+
* @param config Transport configuration.
|
|
65070
|
+
* @returns A transport that executes one leg through the proxy.
|
|
65071
|
+
*/
|
|
65072
|
+
function createGatewayTransport(config) {
|
|
65073
|
+
const doFetch = config.fetchImpl ?? fetch;
|
|
65074
|
+
return {
|
|
65075
|
+
name: "gateway",
|
|
65076
|
+
async execute(request) {
|
|
65077
|
+
const messages = buildMessages(request);
|
|
65078
|
+
const body = {
|
|
65079
|
+
model: config.modelNameFor(request),
|
|
65080
|
+
messages,
|
|
65081
|
+
...request.params,
|
|
65082
|
+
};
|
|
65083
|
+
let response;
|
|
65084
|
+
try {
|
|
65085
|
+
response = await doFetch(`${config.baseUrl.replace(/\/$/, "")}/chat/completions`, {
|
|
65086
|
+
method: "POST",
|
|
65087
|
+
headers: {
|
|
65088
|
+
"content-type": "application/json",
|
|
65089
|
+
authorization: `Bearer ${readGatewayKey(config.apiKeyEnv)}`,
|
|
65090
|
+
...(request.correlationId === undefined
|
|
65091
|
+
? {}
|
|
65092
|
+
: { "x-correlation-id": request.correlationId }),
|
|
65093
|
+
},
|
|
65094
|
+
body: JSON.stringify(body),
|
|
65095
|
+
signal: request.signal,
|
|
65096
|
+
});
|
|
65097
|
+
}
|
|
65098
|
+
catch (error) {
|
|
65099
|
+
// A transport-level throw means the proxy was never reached. Abort is
|
|
65100
|
+
// re-thrown untouched so the chain's timeout classification stays
|
|
65101
|
+
// accurate rather than being masked as a gateway outage.
|
|
65102
|
+
if (request.signal.aborted) {
|
|
65103
|
+
throw error;
|
|
65104
|
+
}
|
|
65105
|
+
throw new GatewayUnreachableError(config.baseUrl, error);
|
|
65106
|
+
}
|
|
65107
|
+
if (!response.ok) {
|
|
65108
|
+
throw new GatewayResponseError(response.status, await response.text());
|
|
65109
|
+
}
|
|
65110
|
+
const payload = (await response.json());
|
|
65111
|
+
const choices = payload.choices;
|
|
65112
|
+
const message = choices?.[0]?.message;
|
|
65113
|
+
return {
|
|
65114
|
+
response: parseContent(message?.content, request.responseFormat),
|
|
65115
|
+
usage: readUsage(payload, request),
|
|
65116
|
+
tool_calls: Array.isArray(message?.tool_calls)
|
|
65117
|
+
? message.tool_calls
|
|
65118
|
+
: undefined,
|
|
65119
|
+
};
|
|
65120
|
+
},
|
|
65121
|
+
};
|
|
65122
|
+
}
|
|
65123
|
+
/**
|
|
65124
|
+
* Compose the message array for a request.
|
|
65125
|
+
*
|
|
65126
|
+
* @param request The request.
|
|
65127
|
+
* @returns Chat messages.
|
|
65128
|
+
*/
|
|
65129
|
+
function buildMessages(request) {
|
|
65130
|
+
const content = typeof request.content === "string" ? request.content : [...request.content];
|
|
65131
|
+
return [{ role: "user", content }];
|
|
65132
|
+
}
|
|
65133
|
+
/**
|
|
65134
|
+
* Interpret the model's content according to the requested format.
|
|
65135
|
+
*
|
|
65136
|
+
* A JSON format that does not parse is an error, not an empty object. Returning
|
|
65137
|
+
* a default here would hand the caller a well-typed value that means nothing,
|
|
65138
|
+
* and the failure would surface much later as a decision made on absent data.
|
|
65139
|
+
*
|
|
65140
|
+
* @param content The raw content.
|
|
65141
|
+
* @param responseFormat The format the caller asked for.
|
|
65142
|
+
* @returns The parsed value.
|
|
65143
|
+
*/
|
|
65144
|
+
function parseContent(content, responseFormat) {
|
|
65145
|
+
const text = typeof content === "string" ? content : "";
|
|
65146
|
+
if (responseFormat === "text") {
|
|
65147
|
+
return text;
|
|
65148
|
+
}
|
|
65149
|
+
try {
|
|
65150
|
+
return JSON.parse(text);
|
|
65151
|
+
}
|
|
65152
|
+
catch (error) {
|
|
65153
|
+
throw new Error(`LLM returned content that is not valid JSON for a ${typeof responseFormat === "string" ? responseFormat : "json_schema"} request: ${error instanceof Error ? error.message : String(error)}`);
|
|
65154
|
+
}
|
|
65155
|
+
}
|
|
65156
|
+
|
|
65157
|
+
/**
|
|
65158
|
+
* The alias-resolving LLM client.
|
|
65159
|
+
*
|
|
65160
|
+
* This is the only supported way to reach a language model from application
|
|
65161
|
+
* code. Its signature mirrors the incumbent client's — content, response
|
|
65162
|
+
* format, options — so migrating a call site is replacing a model string with
|
|
65163
|
+
* an alias, and nothing else. That similarity is the point: a migration that
|
|
65164
|
+
* required rewriting call sites would be a migration that stalls half-done,
|
|
65165
|
+
* leaving some calls inside the timeout, breaker and fallback controls and
|
|
65166
|
+
* some outside them, which is worse than either end state.
|
|
65167
|
+
*
|
|
65168
|
+
* There is deliberately no way to name a model. PD-5 makes vendor strings a CI
|
|
65169
|
+
* failure in application code, and an option that accepted one would let a call
|
|
65170
|
+
* site opt out of the routing policy without anyone noticing.
|
|
65171
|
+
*
|
|
65172
|
+
* Every call gets, in order: alias resolution against the canonical route
|
|
65173
|
+
* table, per-provider parameter normalisation, a hard per-leg timeout, a
|
|
65174
|
+
* per-route circuit breaker, an ordered fallback chain ending at the closed
|
|
65175
|
+
* incumbent, and — where the caller supplies a validator — one schema-feedback
|
|
65176
|
+
* retry ahead of the chain. None of them is optional, because a control that a
|
|
65177
|
+
* caller can switch off is a control that will be off on the call that needed
|
|
65178
|
+
* it.
|
|
65179
|
+
*
|
|
65180
|
+
* @module llm/alias-client
|
|
65181
|
+
*/
|
|
65182
|
+
/** Env var naming the gateway's base URL. */
|
|
65183
|
+
const GATEWAY_BASE_URL_ENV = "LLM_GATEWAY_BASE_URL";
|
|
65184
|
+
/** Env var NAME holding the gateway key. The key itself is never read here. */
|
|
65185
|
+
const DEFAULT_GATEWAY_KEY_ENV = "LLM_GATEWAY_API_KEY";
|
|
65186
|
+
/** Process-wide breaker registry, so route health is shared across call sites. */
|
|
65187
|
+
let breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker);
|
|
65188
|
+
/** Active runtime wiring. */
|
|
65189
|
+
let config = {};
|
|
65190
|
+
/** Lazily built transports, rebuilt whenever configuration changes. */
|
|
65191
|
+
let gatewayTransport = null;
|
|
65192
|
+
let directTransport = null;
|
|
65193
|
+
/**
|
|
65194
|
+
* Wire the client.
|
|
65195
|
+
*
|
|
65196
|
+
* Called once at process start. Transports are injectable so a consumer can
|
|
65197
|
+
* supply its own instrumented client, and so tests can exercise the chain
|
|
65198
|
+
* without a network — a fallback chain that could only be observed against live
|
|
65199
|
+
* providers would in practice never be observed at all.
|
|
65200
|
+
*
|
|
65201
|
+
* @param next Runtime wiring; unspecified fields fall back to the environment.
|
|
65202
|
+
* @returns void
|
|
65203
|
+
*/
|
|
65204
|
+
function configureLlmClient(next) {
|
|
65205
|
+
config = { ...next };
|
|
65206
|
+
gatewayTransport = null;
|
|
65207
|
+
directTransport = null;
|
|
65208
|
+
breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker, next.now);
|
|
65209
|
+
}
|
|
65210
|
+
/**
|
|
65211
|
+
* Inspect route health.
|
|
65212
|
+
*
|
|
65213
|
+
* @returns The live breaker registry.
|
|
65214
|
+
*/
|
|
65215
|
+
function llmBreakers() {
|
|
65216
|
+
return breakers;
|
|
65217
|
+
}
|
|
65218
|
+
/**
|
|
65219
|
+
* Resolve the gateway transport, building it on first use.
|
|
65220
|
+
*
|
|
65221
|
+
* @param chain The chain being served, used to derive each leg's gateway name.
|
|
65222
|
+
* @returns The transport, or null when no gateway is configured.
|
|
65223
|
+
*/
|
|
65224
|
+
function gatewayFor(chain) {
|
|
65225
|
+
if (config.gatewayTransport !== undefined) {
|
|
65226
|
+
return config.gatewayTransport;
|
|
65227
|
+
}
|
|
65228
|
+
const baseUrl = config.gatewayBaseUrl ?? process.env[GATEWAY_BASE_URL_ENV];
|
|
65229
|
+
if (baseUrl === undefined || baseUrl.length === 0) {
|
|
65230
|
+
return null;
|
|
65231
|
+
}
|
|
65232
|
+
if (gatewayTransport === null) {
|
|
65233
|
+
gatewayTransport = createGatewayTransport({
|
|
65234
|
+
baseUrl,
|
|
65235
|
+
apiKeyEnv: config.gatewayApiKeyEnv ?? DEFAULT_GATEWAY_KEY_ENV,
|
|
65236
|
+
modelNameFor: (request) => gatewayModelNameFor(request.route, chain),
|
|
65237
|
+
});
|
|
65238
|
+
}
|
|
65239
|
+
return gatewayTransport;
|
|
65240
|
+
}
|
|
65241
|
+
/**
|
|
65242
|
+
* Resolve the degraded direct transport, building it on first use.
|
|
65243
|
+
*
|
|
65244
|
+
* @returns The transport.
|
|
65245
|
+
*/
|
|
65246
|
+
function directFor() {
|
|
65247
|
+
if (config.directTransport !== undefined) {
|
|
65248
|
+
return config.directTransport;
|
|
65249
|
+
}
|
|
65250
|
+
if (directTransport === null) {
|
|
65251
|
+
directTransport = createDirectTransport({
|
|
65252
|
+
resolveCaller: resolveDefaultDirectCaller,
|
|
65253
|
+
});
|
|
65254
|
+
}
|
|
65255
|
+
return directTransport;
|
|
65256
|
+
}
|
|
65257
|
+
/**
|
|
65258
|
+
* Prepare each leg, normalising parameters up front.
|
|
65259
|
+
*
|
|
65260
|
+
* Normalisation happens before the walk rather than inside it so a capability
|
|
65261
|
+
* mismatch is known without spending a network round trip on it, and so a leg
|
|
65262
|
+
* that cannot serve the request is recorded as skipped rather than counted
|
|
65263
|
+
* against the provider's health.
|
|
65264
|
+
*
|
|
65265
|
+
* @param chain The resolved chain.
|
|
65266
|
+
* @param options The caller's options.
|
|
65267
|
+
* @param responseFormat The requested response shape.
|
|
65268
|
+
* @param transport The transport to carry every leg.
|
|
65269
|
+
* @returns Prepared legs, in chain order.
|
|
65270
|
+
*/
|
|
65271
|
+
function prepareLegs(chain, options, responseFormat, transport) {
|
|
65272
|
+
return chain.routes.map((route) => {
|
|
65273
|
+
try {
|
|
65274
|
+
return {
|
|
65275
|
+
route,
|
|
65276
|
+
transport,
|
|
65277
|
+
params: normaliseParams(options, route, responseFormat),
|
|
65278
|
+
};
|
|
65279
|
+
}
|
|
65280
|
+
catch (error) {
|
|
65281
|
+
if (error instanceof UnsupportedCapabilityError) {
|
|
65282
|
+
return { route, transport, params: error };
|
|
65283
|
+
}
|
|
65284
|
+
throw error;
|
|
65285
|
+
}
|
|
65286
|
+
});
|
|
65287
|
+
}
|
|
65288
|
+
/**
|
|
65289
|
+
* Call a language model by semantic alias.
|
|
65290
|
+
*
|
|
65291
|
+
* @param content The prompt, or multi-part content for a vision-capable route.
|
|
65292
|
+
* @param responseFormat The response shape. Defaults to plain text.
|
|
65293
|
+
* @param options Call options; `alias` is required and there is no model option.
|
|
65294
|
+
* @returns The answer, with the routing decision and full attempt record attached.
|
|
65295
|
+
* @throws {UnknownAliasError} When the alias is not in the route table.
|
|
65296
|
+
* @throws {ChainExhaustedError} When no leg produced an answer.
|
|
65297
|
+
* @throws {SchemaRetryExhaustedError} When a validated payload failed twice.
|
|
65298
|
+
*/
|
|
65299
|
+
async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
65300
|
+
const chain = resolveChain(options.alias, {
|
|
65301
|
+
isolated: options.isolated,
|
|
65302
|
+
timeoutMsOverride: options.timeoutMs,
|
|
65303
|
+
});
|
|
65304
|
+
if (chain.routes.length === 0) {
|
|
65305
|
+
// Nothing is servable. The exclusions say why each leg was unavailable,
|
|
65306
|
+
// which is the difference between an operator reading config and an
|
|
65307
|
+
// operator reading the error.
|
|
65308
|
+
throw new ChainExhaustedError(options.alias, chain.exclusions.map((exclusion) => ({
|
|
65309
|
+
routeKey: `${options.alias}#${exclusion.role}`,
|
|
65310
|
+
role: exclusion.role,
|
|
65311
|
+
provider: exclusion.provider,
|
|
65312
|
+
modelId: "(unresolved)",
|
|
65313
|
+
outcome: "skipped",
|
|
65314
|
+
durationMs: 0,
|
|
65315
|
+
reason: exclusion.reason,
|
|
65316
|
+
})), {
|
|
65317
|
+
prompt_tokens: 0,
|
|
65318
|
+
completion_tokens: 0,
|
|
65319
|
+
provider: "none",
|
|
65320
|
+
model: "none",
|
|
65321
|
+
cost: 0,
|
|
65322
|
+
});
|
|
65323
|
+
}
|
|
65324
|
+
const gateway = gatewayFor(chain);
|
|
65325
|
+
const attemptLog = [];
|
|
65326
|
+
/**
|
|
65327
|
+
* Run the chain, falling back from the gateway to the degraded direct path
|
|
65328
|
+
* only when the gateway itself is unreachable.
|
|
65329
|
+
*
|
|
65330
|
+
* @param prompt The prompt for this attempt.
|
|
65331
|
+
* @returns The transport response and the routing facts about it.
|
|
65332
|
+
*/
|
|
65333
|
+
const runOnce = async (prompt) => {
|
|
65334
|
+
const boundContent = prompt;
|
|
65335
|
+
if (gateway !== null) {
|
|
65336
|
+
try {
|
|
65337
|
+
const outcome = await executeChain(options.alias, {
|
|
65338
|
+
legs: prepareLegs(chain, options, responseFormat, gateway),
|
|
65339
|
+
content: boundContent,
|
|
65340
|
+
responseFormat,
|
|
65341
|
+
breakers,
|
|
65342
|
+
correlationId: options.correlationId,
|
|
65343
|
+
callerSignal: options.signal,
|
|
65344
|
+
now: config.now,
|
|
65345
|
+
onAttempt: (record) => attemptLog.push(record),
|
|
65346
|
+
});
|
|
65347
|
+
return { ...outcome, degraded: false };
|
|
65348
|
+
}
|
|
65349
|
+
catch (error) {
|
|
65350
|
+
if (!isGatewayOutage(error)) {
|
|
65351
|
+
throw error;
|
|
65352
|
+
}
|
|
65353
|
+
// The proxy is gone, not a provider behind it. Walking the chain again
|
|
65354
|
+
// through the gateway would repeat the same failure on every leg, so
|
|
65355
|
+
// the degraded path takes over — restricted to the closed incumbent.
|
|
65356
|
+
}
|
|
65357
|
+
}
|
|
65358
|
+
const direct = directFor();
|
|
65359
|
+
const closedLegs = chain.routes.filter((route) => route.provider.tier === "closed");
|
|
65360
|
+
if (closedLegs.length === 0) {
|
|
65361
|
+
throw new ChainExhaustedError(options.alias, attemptLog, {
|
|
65362
|
+
prompt_tokens: 0,
|
|
65363
|
+
completion_tokens: 0,
|
|
65364
|
+
provider: "none",
|
|
65365
|
+
model: "none",
|
|
65366
|
+
cost: 0,
|
|
65367
|
+
});
|
|
65368
|
+
}
|
|
65369
|
+
const outcome = await executeChain(options.alias, {
|
|
65370
|
+
legs: prepareLegs({ ...chain, routes: closedLegs }, options, responseFormat, direct),
|
|
65371
|
+
content: boundContent,
|
|
65372
|
+
responseFormat,
|
|
65373
|
+
breakers,
|
|
65374
|
+
correlationId: options.correlationId,
|
|
65375
|
+
callerSignal: options.signal,
|
|
65376
|
+
now: config.now,
|
|
65377
|
+
onAttempt: (record) => attemptLog.push(record),
|
|
65378
|
+
});
|
|
65379
|
+
return { ...outcome, degraded: true };
|
|
65380
|
+
};
|
|
65381
|
+
if (options.validate === undefined) {
|
|
65382
|
+
const outcome = await runOnce(content);
|
|
65383
|
+
return {
|
|
65384
|
+
response: outcome.response.response,
|
|
65385
|
+
usage: outcome.response.usage,
|
|
65386
|
+
tool_calls: outcome.response.tool_calls,
|
|
65387
|
+
servedBy: outcome.servedBy,
|
|
65388
|
+
attempts: attemptLog,
|
|
65389
|
+
degraded: outcome.degraded,
|
|
65390
|
+
totalUsage: outcome.totalUsage,
|
|
65391
|
+
};
|
|
65392
|
+
}
|
|
65393
|
+
// A validator is only meaningful against a text prompt, because the retry has
|
|
65394
|
+
// to be able to append the validator's complaint to it.
|
|
65395
|
+
if (typeof content !== "string") {
|
|
65396
|
+
throw new Error("a validator requires a string prompt: the feedback retry appends the validator's rejection to the original prompt");
|
|
65397
|
+
}
|
|
65398
|
+
let lastRouting = null;
|
|
65399
|
+
const validated = await callWithValidation({
|
|
65400
|
+
prompt: content,
|
|
65401
|
+
validate: options.validate,
|
|
65402
|
+
call: async (prompt) => {
|
|
65403
|
+
const outcome = await runOnce(prompt);
|
|
65404
|
+
lastRouting = {
|
|
65405
|
+
servedBy: outcome.servedBy,
|
|
65406
|
+
degraded: outcome.degraded,
|
|
65407
|
+
totalUsage: outcome.totalUsage,
|
|
65408
|
+
};
|
|
65409
|
+
return outcome.response;
|
|
65410
|
+
},
|
|
65411
|
+
});
|
|
65412
|
+
if (lastRouting === null) {
|
|
65413
|
+
throw new Error("validated call completed without recording a routing decision");
|
|
65414
|
+
}
|
|
65415
|
+
const routing = lastRouting;
|
|
65416
|
+
return {
|
|
65417
|
+
response: validated.value,
|
|
65418
|
+
usage: validated.response.usage,
|
|
65419
|
+
tool_calls: validated.response.tool_calls,
|
|
65420
|
+
servedBy: routing.servedBy,
|
|
65421
|
+
attempts: attemptLog,
|
|
65422
|
+
degraded: routing.degraded,
|
|
65423
|
+
totalUsage: validated.totalUsage,
|
|
65424
|
+
};
|
|
65425
|
+
}
|
|
65426
|
+
/**
|
|
65427
|
+
* Whether an error means the gateway itself is gone.
|
|
65428
|
+
*
|
|
65429
|
+
* Only a transport-level failure to reach the proxy qualifies. A provider error
|
|
65430
|
+
* relayed BY the proxy is a normal chain event and must not trigger the
|
|
65431
|
+
* degraded path, or a single flaky provider would quietly move every call onto
|
|
65432
|
+
* the closed incumbent — a fallback the routing policy reserves for last.
|
|
65433
|
+
*
|
|
65434
|
+
* @param error The error to classify.
|
|
65435
|
+
* @returns Whether the gateway is unreachable.
|
|
65436
|
+
*/
|
|
65437
|
+
function isGatewayOutage(error) {
|
|
65438
|
+
if (error instanceof GatewayUnreachableError) {
|
|
65439
|
+
return true;
|
|
65440
|
+
}
|
|
65441
|
+
if (error instanceof ChainExhaustedError) {
|
|
65442
|
+
return (error.attempts.length > 0 &&
|
|
65443
|
+
error.attempts.every((attempt) => attempt.reason === undefined
|
|
65444
|
+
? false
|
|
65445
|
+
: attempt.reason.includes("is unreachable")));
|
|
65446
|
+
}
|
|
65447
|
+
return false;
|
|
65448
|
+
}
|
|
65449
|
+
/**
|
|
65450
|
+
* The aliases application code may name.
|
|
65451
|
+
*
|
|
65452
|
+
* @returns The alias names, sorted.
|
|
65453
|
+
*/
|
|
65454
|
+
function llmAliases() {
|
|
65455
|
+
return Object.keys(routeTable.aliases).sort();
|
|
65456
|
+
}
|
|
65457
|
+
|
|
65458
|
+
/**
|
|
65459
|
+
* Streaming normalisation across provider wire formats.
|
|
65460
|
+
*
|
|
65461
|
+
* A caller that streams wants one thing — text as it arrives — but the wire
|
|
65462
|
+
* formats disagree about how to say it. OpenAI-compatible providers emit
|
|
65463
|
+
* server-sent events whose payload nests the increment under
|
|
65464
|
+
* `choices[0].delta.content` and end with a literal `[DONE]` sentinel;
|
|
65465
|
+
* Anthropic emits typed events where the increment is `delta.text` and the end
|
|
65466
|
+
* is an explicit `message_stop`. A consumer written against one shape breaks on
|
|
65467
|
+
* the other, which would make a fallback across providers fail precisely when
|
|
65468
|
+
* the fallback was needed.
|
|
65469
|
+
*
|
|
65470
|
+
* The one behaviour that matters more than the format is how a stream ENDS. A
|
|
65471
|
+
* stream cut short mid-answer looks exactly like a short answer: the consumer
|
|
65472
|
+
* has already received and probably already acted on the text. So a stream
|
|
65473
|
+
* that stops without its terminal event raises rather than returning what it
|
|
65474
|
+
* had — a truncated answer accepted as complete is a wrong answer that nothing
|
|
65475
|
+
* reports.
|
|
65476
|
+
*
|
|
65477
|
+
* @module llm/streaming
|
|
65478
|
+
*/
|
|
65479
|
+
/** SSE payload that marks the end of an OpenAI-compatible stream. */
|
|
65480
|
+
const SSE_DONE_SENTINEL = "[DONE]";
|
|
65481
|
+
/** Prefix carrying the payload on an SSE line. */
|
|
65482
|
+
const SSE_DATA_PREFIX = "data:";
|
|
65483
|
+
/** Anthropic event type carrying a text increment. */
|
|
65484
|
+
const ANTHROPIC_DELTA_EVENT = "content_block_delta";
|
|
65485
|
+
/** Anthropic event type marking the end of a message. */
|
|
65486
|
+
const ANTHROPIC_STOP_EVENT = "message_stop";
|
|
65487
|
+
/** Anthropic event type carrying a mid-stream error. */
|
|
65488
|
+
const ANTHROPIC_ERROR_EVENT = "error";
|
|
65489
|
+
/**
|
|
65490
|
+
* Thrown when a stream ends without its terminal event.
|
|
65491
|
+
*
|
|
65492
|
+
* A distinct type because the caller's correct response differs from a normal
|
|
65493
|
+
* failure: the partial text exists and may be worth logging for diagnosis, but
|
|
65494
|
+
* it must never be treated as the answer.
|
|
65495
|
+
*/
|
|
65496
|
+
class StreamTruncatedError extends Error {
|
|
65497
|
+
/** Text received before the stream stopped. Present for diagnosis only. */
|
|
65498
|
+
partialText;
|
|
65499
|
+
/**
|
|
65500
|
+
* @param partialText What had arrived when the stream stopped.
|
|
65501
|
+
* @param reason Why it stopped, when known.
|
|
65502
|
+
*/
|
|
65503
|
+
constructor(partialText, reason) {
|
|
65504
|
+
super(`LLM stream ended without a terminal event after ${partialText.length} characters: ${reason}. ` +
|
|
65505
|
+
"The partial text is not returned as an answer: a truncated answer accepted as complete is a wrong answer nothing reports.");
|
|
65506
|
+
this.name = "StreamTruncatedError";
|
|
65507
|
+
this.partialText = partialText;
|
|
65508
|
+
}
|
|
65509
|
+
}
|
|
65510
|
+
/** Thrown when a provider reports an error inside an already-open stream. */
|
|
65511
|
+
class StreamProviderError extends Error {
|
|
65512
|
+
/** Text received before the error. */
|
|
65513
|
+
partialText;
|
|
65514
|
+
/**
|
|
65515
|
+
* @param partialText What had arrived when the error appeared.
|
|
65516
|
+
* @param detail The provider's message.
|
|
65517
|
+
*/
|
|
65518
|
+
constructor(partialText, detail) {
|
|
65519
|
+
super(`LLM stream failed mid-response: ${detail}`);
|
|
65520
|
+
this.name = "StreamProviderError";
|
|
65521
|
+
this.partialText = partialText;
|
|
65522
|
+
}
|
|
65523
|
+
}
|
|
65524
|
+
/**
|
|
65525
|
+
* Split a byte stream into complete lines.
|
|
65526
|
+
*
|
|
65527
|
+
* Chunk boundaries fall wherever the network puts them, not on line breaks, so
|
|
65528
|
+
* a partial line is carried across chunks. Parsing each network chunk as if it
|
|
65529
|
+
* were whole would drop or corrupt every event unlucky enough to be split.
|
|
65530
|
+
*
|
|
65531
|
+
* @param source The response body stream.
|
|
65532
|
+
* @returns An async iterable of complete lines.
|
|
65533
|
+
*/
|
|
65534
|
+
async function* toLines(source) {
|
|
65535
|
+
const decoder = new TextDecoder();
|
|
65536
|
+
let buffer = "";
|
|
65537
|
+
for await (const chunk of source) {
|
|
65538
|
+
buffer += decoder.decode(chunk, { stream: true });
|
|
65539
|
+
let newline = buffer.indexOf("\n");
|
|
65540
|
+
while (newline !== -1) {
|
|
65541
|
+
yield buffer.slice(0, newline).replace(/\r$/, "");
|
|
65542
|
+
buffer = buffer.slice(newline + 1);
|
|
65543
|
+
newline = buffer.indexOf("\n");
|
|
65544
|
+
}
|
|
65545
|
+
}
|
|
65546
|
+
buffer += decoder.decode();
|
|
65547
|
+
if (buffer.length > 0) {
|
|
65548
|
+
yield buffer;
|
|
65549
|
+
}
|
|
65550
|
+
}
|
|
65551
|
+
/**
|
|
65552
|
+
* Normalise an OpenAI-compatible SSE stream.
|
|
65553
|
+
*
|
|
65554
|
+
* @param source The response body stream.
|
|
65555
|
+
* @returns Normalised chunks.
|
|
65556
|
+
* @throws {StreamTruncatedError} When the stream ends without `[DONE]`.
|
|
65557
|
+
* @throws {StreamProviderError} When an error event appears mid-stream.
|
|
65558
|
+
*/
|
|
65559
|
+
async function* normaliseOpenAiStream(source) {
|
|
65560
|
+
let text = "";
|
|
65561
|
+
let sawDone = false;
|
|
65562
|
+
for await (const line of toLines(source)) {
|
|
65563
|
+
if (!line.startsWith(SSE_DATA_PREFIX)) {
|
|
65564
|
+
continue;
|
|
65565
|
+
}
|
|
65566
|
+
const payload = line.slice(SSE_DATA_PREFIX.length).trim();
|
|
65567
|
+
if (payload === SSE_DONE_SENTINEL) {
|
|
65568
|
+
sawDone = true;
|
|
65569
|
+
break;
|
|
65570
|
+
}
|
|
65571
|
+
if (payload.length === 0) {
|
|
65572
|
+
continue;
|
|
65573
|
+
}
|
|
65574
|
+
let event;
|
|
65575
|
+
try {
|
|
65576
|
+
event = JSON.parse(payload);
|
|
65577
|
+
}
|
|
65578
|
+
catch (error) {
|
|
65579
|
+
throw new StreamProviderError(text, `unparseable event: ${error instanceof Error ? error.message : String(error)}`);
|
|
65580
|
+
}
|
|
65581
|
+
if (event.error !== undefined) {
|
|
65582
|
+
throw new StreamProviderError(text, JSON.stringify(event.error));
|
|
65583
|
+
}
|
|
65584
|
+
const choices = event.choices;
|
|
65585
|
+
const delta = choices?.[0]?.delta?.content;
|
|
65586
|
+
if (typeof delta === "string" && delta.length > 0) {
|
|
65587
|
+
text += delta;
|
|
65588
|
+
yield { delta, text };
|
|
65589
|
+
}
|
|
65590
|
+
}
|
|
65591
|
+
if (!sawDone) {
|
|
65592
|
+
throw new StreamTruncatedError(text, "no [DONE] sentinel was received");
|
|
65593
|
+
}
|
|
65594
|
+
}
|
|
65595
|
+
/**
|
|
65596
|
+
* Normalise an Anthropic event stream.
|
|
65597
|
+
*
|
|
65598
|
+
* @param source The response body stream.
|
|
65599
|
+
* @returns Normalised chunks.
|
|
65600
|
+
* @throws {StreamTruncatedError} When the stream ends without `message_stop`.
|
|
65601
|
+
* @throws {StreamProviderError} When an error event appears mid-stream.
|
|
65602
|
+
*/
|
|
65603
|
+
async function* normaliseAnthropicStream(source) {
|
|
65604
|
+
let text = "";
|
|
65605
|
+
let sawStop = false;
|
|
65606
|
+
for await (const line of toLines(source)) {
|
|
65607
|
+
if (!line.startsWith(SSE_DATA_PREFIX)) {
|
|
65608
|
+
continue;
|
|
65609
|
+
}
|
|
65610
|
+
const payload = line.slice(SSE_DATA_PREFIX.length).trim();
|
|
65611
|
+
if (payload.length === 0) {
|
|
65612
|
+
continue;
|
|
65613
|
+
}
|
|
65614
|
+
let event;
|
|
65615
|
+
try {
|
|
65616
|
+
event = JSON.parse(payload);
|
|
65617
|
+
}
|
|
65618
|
+
catch (error) {
|
|
65619
|
+
throw new StreamProviderError(text, `unparseable event: ${error instanceof Error ? error.message : String(error)}`);
|
|
65620
|
+
}
|
|
65621
|
+
const type = event.type;
|
|
65622
|
+
if (type === ANTHROPIC_ERROR_EVENT) {
|
|
65623
|
+
throw new StreamProviderError(text, JSON.stringify(event.error ?? event));
|
|
65624
|
+
}
|
|
65625
|
+
if (type === ANTHROPIC_STOP_EVENT) {
|
|
65626
|
+
sawStop = true;
|
|
65627
|
+
break;
|
|
65628
|
+
}
|
|
65629
|
+
if (type === ANTHROPIC_DELTA_EVENT) {
|
|
65630
|
+
const delta = event.delta?.text;
|
|
65631
|
+
if (typeof delta === "string" && delta.length > 0) {
|
|
65632
|
+
text += delta;
|
|
65633
|
+
yield { delta, text };
|
|
65634
|
+
}
|
|
65635
|
+
}
|
|
65636
|
+
}
|
|
65637
|
+
if (!sawStop) {
|
|
65638
|
+
throw new StreamTruncatedError(text, "no message_stop event was received");
|
|
65639
|
+
}
|
|
65640
|
+
}
|
|
65641
|
+
/**
|
|
65642
|
+
* Normalise a stream according to the wire format its provider speaks.
|
|
65643
|
+
*
|
|
65644
|
+
* Selecting on the route's declared API style rather than sniffing the payload
|
|
65645
|
+
* keeps the decision with the route table, which is the thing that actually
|
|
65646
|
+
* knows which provider is answering.
|
|
65647
|
+
*
|
|
65648
|
+
* @param apiStyle The provider's wire format.
|
|
65649
|
+
* @param source The response body stream.
|
|
65650
|
+
* @returns Normalised chunks, identical in shape across providers.
|
|
65651
|
+
*/
|
|
65652
|
+
function normaliseStream(apiStyle, source) {
|
|
65653
|
+
return apiStyle === "anthropic"
|
|
65654
|
+
? normaliseAnthropicStream(source)
|
|
65655
|
+
: normaliseOpenAiStream(source);
|
|
65656
|
+
}
|
|
65657
|
+
/**
|
|
65658
|
+
* Drain a normalised stream into its complete text.
|
|
65659
|
+
*
|
|
65660
|
+
* @param stream A normalised stream.
|
|
65661
|
+
* @returns The full text.
|
|
65662
|
+
* @throws Whatever the stream raises; a truncated stream never resolves to text.
|
|
65663
|
+
*/
|
|
65664
|
+
async function collectStream(stream) {
|
|
65665
|
+
let text = "";
|
|
65666
|
+
for await (const chunk of stream) {
|
|
65667
|
+
text = chunk.text;
|
|
65668
|
+
}
|
|
65669
|
+
return text;
|
|
65670
|
+
}
|
|
65671
|
+
|
|
62914
65672
|
/**
|
|
62915
65673
|
* @module LRUCache
|
|
62916
65674
|
*/
|
|
@@ -72035,7 +74793,7 @@ const adaptic = {
|
|
|
72035
74793
|
},
|
|
72036
74794
|
rateLimiter: {
|
|
72037
74795
|
TokenBucketRateLimiter,
|
|
72038
|
-
limiters: rateLimiters,
|
|
74796
|
+
limiters: rateLimiters$1,
|
|
72039
74797
|
},
|
|
72040
74798
|
};
|
|
72041
74799
|
const adptc = adaptic;
|
|
@@ -72070,6 +74828,8 @@ exports.AssetAllocationEngine = AssetAllocationEngine;
|
|
|
72070
74828
|
exports.AuthenticationError = AuthenticationError;
|
|
72071
74829
|
exports.BTC_PAIRS = BTC_PAIRS;
|
|
72072
74830
|
exports.BarError = BarError;
|
|
74831
|
+
exports.ChainExhaustedError = ChainExhaustedError;
|
|
74832
|
+
exports.CircuitBreakerRegistry = CircuitBreakerRegistry;
|
|
72073
74833
|
exports.CircuitOpenError = CircuitOpenError;
|
|
72074
74834
|
exports.CryptoDataError = CryptoDataError;
|
|
72075
74835
|
exports.CryptoOrderError = CryptoOrderError;
|
|
@@ -72078,7 +74838,10 @@ exports.DEFAULT_RISK_FREE_RATE = DEFAULT_RISK_FREE_RATE;
|
|
|
72078
74838
|
exports.DEFAULT_TIMEOUTS = DEFAULT_TIMEOUTS;
|
|
72079
74839
|
exports.DEFAULT_TRADING_POLICY = DEFAULT_TRADING_POLICY;
|
|
72080
74840
|
exports.DataFormatError = DataFormatError;
|
|
74841
|
+
exports.DirectTransportRefusedError = DirectTransportRefusedError;
|
|
72081
74842
|
exports.DuplicateClientOrderIdError = DuplicateClientOrderIdError;
|
|
74843
|
+
exports.GatewayResponseError = GatewayResponseError;
|
|
74844
|
+
exports.GatewayUnreachableError = GatewayUnreachableError;
|
|
72082
74845
|
exports.HttpClientError = HttpClientError;
|
|
72083
74846
|
exports.HttpServerError = HttpServerError;
|
|
72084
74847
|
exports.KEEP_ALIVE_DEFAULTS = KEEP_ALIVE_DEFAULTS;
|
|
@@ -72095,13 +74858,18 @@ exports.MassiveTradeZodSchema = MassiveTradeSchema;
|
|
|
72095
74858
|
exports.MassiveTradesResponseSchema = MassiveTradesResponseSchema;
|
|
72096
74859
|
exports.NetworkError = NetworkError;
|
|
72097
74860
|
exports.NewsError = NewsError;
|
|
74861
|
+
exports.NoServableRouteError = NoServableRouteError;
|
|
72098
74862
|
exports.OptionStrategyError = OptionStrategyError;
|
|
72099
74863
|
exports.OptionsDataError = OptionsDataError;
|
|
72100
74864
|
exports.QuoteError = QuoteError;
|
|
72101
74865
|
exports.RISK_FREE_RATE_TTL_MS = RISK_FREE_RATE_TTL_MS;
|
|
74866
|
+
exports.RateGuardTimeoutError = RateGuardTimeoutError;
|
|
72102
74867
|
exports.RateLimitError = RateLimitError;
|
|
72103
74868
|
exports.RawMassivePriceDataSchema = RawMassivePriceDataSchema;
|
|
74869
|
+
exports.SchemaRetryExhaustedError = SchemaRetryExhaustedError;
|
|
72104
74870
|
exports.StampedeProtectedCache = StampedeProtectedCache;
|
|
74871
|
+
exports.StreamProviderError = StreamProviderError;
|
|
74872
|
+
exports.StreamTruncatedError = StreamTruncatedError;
|
|
72105
74873
|
exports.TRADING_API = TRADING_API;
|
|
72106
74874
|
exports.TimeoutError = TimeoutError;
|
|
72107
74875
|
exports.TokenBucketRateLimiter = TokenBucketRateLimiter;
|
|
@@ -72110,7 +74878,9 @@ exports.TrailingStopValidationError = TrailingStopValidationError;
|
|
|
72110
74878
|
exports.USDC_PAIRS = USDC_PAIRS;
|
|
72111
74879
|
exports.USDT_PAIRS = USDT_PAIRS;
|
|
72112
74880
|
exports.USD_PAIRS = USD_PAIRS;
|
|
74881
|
+
exports.UnknownAliasError = UnknownAliasError;
|
|
72113
74882
|
exports.UnsupportedBrokerError = UnsupportedBrokerError;
|
|
74883
|
+
exports.UnsupportedCapabilityError = UnsupportedCapabilityError;
|
|
72114
74884
|
exports.ValidationError = ValidationError;
|
|
72115
74885
|
exports.ValidationResponseError = ValidationResponseError;
|
|
72116
74886
|
exports.WEBSOCKET_STREAMS = WEBSOCKET_STREAMS;
|
|
@@ -72125,6 +74895,7 @@ exports.atr = atrNs;
|
|
|
72125
74895
|
exports.bracketOrders = bracketOrders;
|
|
72126
74896
|
exports.buildOCCSymbol = buildOCCSymbol;
|
|
72127
74897
|
exports.buildOptionSymbol = buildOptionSymbol;
|
|
74898
|
+
exports.buildRetryPrompt = buildRetryPrompt;
|
|
72128
74899
|
exports.buyCryptoNotional = buyCryptoNotional;
|
|
72129
74900
|
exports.buyToClose = buyToClose;
|
|
72130
74901
|
exports.buyToOpen = buyToOpen;
|
|
@@ -72135,6 +74906,8 @@ exports.calculateOrderValue = calculateOrderValue;
|
|
|
72135
74906
|
exports.calculatePeriodPerformance = calculatePeriodPerformance;
|
|
72136
74907
|
exports.calculatePutCallRatio = calculatePutCallRatio;
|
|
72137
74908
|
exports.calculateTotalFilledValue = calculateTotalFilledValue;
|
|
74909
|
+
exports.callLLMByAlias = callLLMByAlias;
|
|
74910
|
+
exports.callWithValidation = callWithValidation;
|
|
72138
74911
|
exports.cancelAllCryptoOrders = cancelAllCryptoOrders;
|
|
72139
74912
|
exports.cancelOCOOrder = cancelOCOOrder;
|
|
72140
74913
|
exports.cancelOTOOrder = cancelOTOOrder;
|
|
@@ -72145,6 +74918,9 @@ exports.clearClientCache = clearClientCache;
|
|
|
72145
74918
|
exports.clock = clock;
|
|
72146
74919
|
exports.closeAllOptionPositions = closeAllOptionPositions;
|
|
72147
74920
|
exports.closeOptionPosition = closeOptionPosition;
|
|
74921
|
+
exports.closedIncumbentLeg = closedIncumbentLeg;
|
|
74922
|
+
exports.collectStream = collectStream;
|
|
74923
|
+
exports.configureLlmClient = configureLlmClient;
|
|
72148
74924
|
exports.createAlpacaClient = createAlpacaClient;
|
|
72149
74925
|
exports.createAlpacaMarketDataAPI = createAlpacaMarketDataAPI;
|
|
72150
74926
|
exports.createAlpacaTradingAPI = createAlpacaTradingAPI;
|
|
@@ -72158,7 +74934,9 @@ exports.createCryptoMarketOrder = createCryptoMarketOrder;
|
|
|
72158
74934
|
exports.createCryptoOrder = createCryptoOrder;
|
|
72159
74935
|
exports.createCryptoStopLimitOrder = createCryptoStopLimitOrder;
|
|
72160
74936
|
exports.createCryptoStopOrder = createCryptoStopOrder;
|
|
74937
|
+
exports.createDirectTransport = createDirectTransport;
|
|
72161
74938
|
exports.createExecutorFromTradingAPI = createExecutorFromTradingAPI;
|
|
74939
|
+
exports.createGatewayTransport = createGatewayTransport;
|
|
72162
74940
|
exports.createIronCondor = createIronCondor$1;
|
|
72163
74941
|
exports.createIronCondorAdvanced = createIronCondor;
|
|
72164
74942
|
exports.createMultiLegOptionOrder = createMultiLegOptionOrder;
|
|
@@ -72192,6 +74970,7 @@ exports.findNearestExpiration = findNearestExpiration;
|
|
|
72192
74970
|
exports.findOptionsByDelta = findOptionsByDelta;
|
|
72193
74971
|
exports.formatOrderForLog = formatOrderForLog;
|
|
72194
74972
|
exports.formatOrderSummary = formatOrderSummary;
|
|
74973
|
+
exports.gatewayModelNameFor = gatewayModelNameFor;
|
|
72195
74974
|
exports.generateOptimalAllocation = generateOptimalAllocation;
|
|
72196
74975
|
exports.getAccountConfiguration = getAccountConfiguration;
|
|
72197
74976
|
exports.getAccountDetails = getAccountDetails;
|
|
@@ -72278,6 +75057,7 @@ exports.getTradingWebSocketUrl = getTradingWebSocketUrl;
|
|
|
72278
75057
|
exports.getTrailingStopHWM = getTrailingStopHWM;
|
|
72279
75058
|
exports.groupOrdersByStatus = groupOrdersByStatus;
|
|
72280
75059
|
exports.groupOrdersBySymbol = groupOrdersBySymbol;
|
|
75060
|
+
exports.guardSnapshots = guardSnapshots;
|
|
72281
75061
|
exports.hasActiveTrailingStop = hasActiveTrailingStop;
|
|
72282
75062
|
exports.hasOptionLiquidity = hasGoodLiquidity;
|
|
72283
75063
|
exports.hasStockLiquidity = hasGoodLiquidity$1;
|
|
@@ -72299,21 +75079,37 @@ exports.isSupportedCryptoPair = isSupportedCryptoPair;
|
|
|
72299
75079
|
exports.isTransientNetworkError = isTransientNetworkError;
|
|
72300
75080
|
exports.legacyApi = index$1;
|
|
72301
75081
|
exports.limitBuyWithTakeProfit = limitBuyWithTakeProfit;
|
|
75082
|
+
exports.limitsFor = limitsFor;
|
|
75083
|
+
exports.limitsInventory = limitsInventory;
|
|
75084
|
+
exports.listAliases = listAliases;
|
|
75085
|
+
exports.llmAliases = llmAliases;
|
|
75086
|
+
exports.llmBreakers = llmBreakers;
|
|
75087
|
+
exports.normaliseAnthropicStream = normaliseAnthropicStream;
|
|
75088
|
+
exports.normaliseOpenAiStream = normaliseOpenAiStream;
|
|
75089
|
+
exports.normaliseParams = normaliseParams;
|
|
75090
|
+
exports.normaliseStream = normaliseStream;
|
|
72302
75091
|
exports.ocoOrders = ocoOrders;
|
|
72303
75092
|
exports.orderUtils = orderUtils;
|
|
75093
|
+
exports.orderedRoutes = orderedRoutes;
|
|
72304
75094
|
exports.otoOrders = otoOrders;
|
|
72305
75095
|
exports.paginate = paginate;
|
|
72306
75096
|
exports.paginateAll = paginateAll;
|
|
72307
75097
|
exports.parseOCCSymbol = parseOCCSymbol;
|
|
72308
75098
|
exports.protectLongPosition = protectLongPosition;
|
|
72309
75099
|
exports.protectShortPosition = protectShortPosition;
|
|
72310
|
-
exports.rateLimiters = rateLimiters;
|
|
75100
|
+
exports.rateLimiters = rateLimiters$1;
|
|
72311
75101
|
exports.resetLogger = resetLogger;
|
|
75102
|
+
exports.resetProviderGuards = resetProviderGuards;
|
|
72312
75103
|
exports.resetRiskFreeRateCache = resetRiskFreeRateCache;
|
|
75104
|
+
exports.resolveChain = resolveChain;
|
|
75105
|
+
exports.resolveDefaultDirectCaller = resolveDefaultDirectCaller;
|
|
72313
75106
|
exports.risk = riskNs;
|
|
72314
75107
|
exports.rollOptionPosition = rollOptionPosition;
|
|
72315
75108
|
exports.roundPriceForAlpaca = roundPriceForAlpaca$3;
|
|
72316
75109
|
exports.roundPriceForAlpacaNumber = roundPriceForAlpacaNumber;
|
|
75110
|
+
exports.routeKeyFor = routeKeyFor;
|
|
75111
|
+
exports.routeSupports = routeSupports;
|
|
75112
|
+
exports.routeTable = routeTable;
|
|
72317
75113
|
exports.safeValidateResponse = safeValidateResponse;
|
|
72318
75114
|
exports.searchNews = searchNews;
|
|
72319
75115
|
exports.sellAllCrypto = sellAllCrypto;
|
|
@@ -72325,6 +75121,7 @@ exports.setRiskFreeRate = setRiskFreeRate;
|
|
|
72325
75121
|
exports.shortWithStopLoss = shortWithStopLoss;
|
|
72326
75122
|
exports.sortOrdersByDate = sortOrdersByDate;
|
|
72327
75123
|
exports.strategy = strategyNs;
|
|
75124
|
+
exports.sumUsage = sumUsage;
|
|
72328
75125
|
exports.tradingPolicy = index;
|
|
72329
75126
|
exports.trailingStops = trailingStops;
|
|
72330
75127
|
exports.updateAccountConfiguration = updateAccountConfiguration;
|
|
@@ -72337,6 +75134,7 @@ exports.validateResponse = validateResponse;
|
|
|
72337
75134
|
exports.verifyFetchKeepAlive = verifyFetchKeepAlive;
|
|
72338
75135
|
exports.volatility = volatilityNs;
|
|
72339
75136
|
exports.waitForOrderFill = waitForOrderFill;
|
|
75137
|
+
exports.withProviderGuards = withProviderGuards;
|
|
72340
75138
|
exports.withRetry = withRetry;
|
|
72341
75139
|
exports.withTimeout = withTimeout;
|
|
72342
75140
|
//# sourceMappingURL=index.cjs.map
|