@adaptic/utils 0.0.1038 → 0.0.1040
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +1784 -345
- package/dist/index.cjs.map +1 -1
- package/dist/index.mjs +1768 -346
- package/dist/index.mjs.map +1 -1
- package/dist/types/__tests__/llm/client/support/legacy-scenarios.d.ts +48 -0
- package/dist/types/__tests__/llm/client/support/legacy-scenarios.d.ts.map +1 -0
- package/dist/types/__tests__/llm/client/support/transports.d.ts +18 -0
- package/dist/types/__tests__/llm/client/support/transports.d.ts.map +1 -1
- package/dist/types/alpaca/index.d.ts.map +1 -1
- package/dist/types/alpaca/trading/index.d.ts +1 -0
- package/dist/types/alpaca/trading/index.d.ts.map +1 -1
- package/dist/types/alpaca/trading/trail-limits.d.ts +8 -0
- package/dist/types/alpaca/trading/trail-limits.d.ts.map +1 -0
- package/dist/types/alpaca/trading/trail-unit.d.ts +98 -0
- package/dist/types/alpaca/trading/trail-unit.d.ts.map +1 -0
- package/dist/types/alpaca/trading/trailing-stops.d.ts +28 -15
- package/dist/types/alpaca/trading/trailing-stops.d.ts.map +1 -1
- package/dist/types/alpaca-trading-api.d.ts.map +1 -1
- package/dist/types/index.d.ts.map +1 -1
- package/dist/types/llm/alias-client.d.ts +10 -1
- package/dist/types/llm/alias-client.d.ts.map +1 -1
- package/dist/types/llm/circuit-breaker.d.ts +64 -2
- package/dist/types/llm/circuit-breaker.d.ts.map +1 -1
- package/dist/types/llm/fallback-chain.d.ts +109 -23
- package/dist/types/llm/fallback-chain.d.ts.map +1 -1
- package/dist/types/llm/hedge.d.ts +107 -0
- package/dist/types/llm/hedge.d.ts.map +1 -0
- package/dist/types/llm/index.d.ts +9 -6
- package/dist/types/llm/index.d.ts.map +1 -1
- package/dist/types/llm/leg-attempt.d.ts +165 -0
- package/dist/types/llm/leg-attempt.d.ts.map +1 -0
- package/dist/types/llm/leg-latency-tracker.d.ts +126 -0
- package/dist/types/llm/leg-latency-tracker.d.ts.map +1 -0
- package/dist/types/llm/rate-guard.d.ts +17 -0
- package/dist/types/llm/rate-guard.d.ts.map +1 -1
- package/dist/types/llm/route-table.d.ts +17 -1
- package/dist/types/llm/route-table.d.ts.map +1 -1
- package/dist/types/llm/transports/gateway.d.ts +27 -2
- package/dist/types/llm/transports/gateway.d.ts.map +1 -1
- package/dist/types/llm/types.d.ts +161 -1
- package/dist/types/llm/types.d.ts.map +1 -1
- package/dist/types/schemas/alpaca-schemas.d.ts +16 -16
- package/dist/types/schemas/massive-schemas.d.ts +10 -10
- package/dist/types/trading-policy/schemas/effective-policy.schema.d.ts +6 -6
- package/dist/types/trading-policy/schemas/policy-mutation.schema.d.ts +12 -12
- package/dist/types/trading-policy/schemas/signal-consumption-prefs.schema.d.ts +8 -8
- package/package.json +1 -1
package/dist/index.cjs
CHANGED
|
@@ -10090,6 +10090,160 @@ class AlpacaMarketDataAPI extends require$$0$4.EventEmitter {
|
|
|
10090
10090
|
// Export the singleton instance
|
|
10091
10091
|
const marketDataAPI = AlpacaMarketDataAPI.getInstance();
|
|
10092
10092
|
|
|
10093
|
+
/**
|
|
10094
|
+
* Alpaca's hard upper limit for `trail_percent` on trailing-stop orders.
|
|
10095
|
+
* Submissions exceeding this value are rejected with HTTP 422 / code 42210000
|
|
10096
|
+
* ("trail_percent must be <= 25"). See:
|
|
10097
|
+
* https://docs.alpaca.markets/reference/postorder
|
|
10098
|
+
*/
|
|
10099
|
+
const ALPACA_MAX_TRAIL_PERCENT = 25;
|
|
10100
|
+
|
|
10101
|
+
/**
|
|
10102
|
+
* Trailing-stop replace-unit contract.
|
|
10103
|
+
*
|
|
10104
|
+
* Alpaca's order replace (`PATCH /v2/orders/{id}`) takes a single unitless
|
|
10105
|
+
* `trail` field. The broker reads it in the unit of the ORIGINAL order: a
|
|
10106
|
+
* `trail_percent` order reads `trail` as a percent, a `trail_price` order
|
|
10107
|
+
* reads it as dollars. A replace cannot change the unit. A caller that holds
|
|
10108
|
+
* a distance in one unit must therefore resolve it against the resting
|
|
10109
|
+
* order's unit before the replace, or the broker stores the number in the
|
|
10110
|
+
* other unit: a $22 dollar distance becomes a 22% trail.
|
|
10111
|
+
*
|
|
10112
|
+
* This module is the pure resolution step. It never guesses a unit, never
|
|
10113
|
+
* defaults a reference price, and refuses rather than send a value the broker
|
|
10114
|
+
* would reject or treat as effectively zero.
|
|
10115
|
+
*/
|
|
10116
|
+
/**
|
|
10117
|
+
* Smallest `trail_percent` this module will send. Below it the trail is a
|
|
10118
|
+
* rounding artefact of the broker's tick grid rather than a protective
|
|
10119
|
+
* distance, so a conversion landing under it is refused.
|
|
10120
|
+
*/
|
|
10121
|
+
const MIN_CONVERTED_TRAIL_PERCENT = 0.1;
|
|
10122
|
+
/** Percent values are sent to the broker at hundredths of a percent. */
|
|
10123
|
+
const PERCENT_DECIMALS_SCALE = 100;
|
|
10124
|
+
/** A ratio expressed in percent. */
|
|
10125
|
+
const PERCENT_PER_UNIT = 100;
|
|
10126
|
+
/**
|
|
10127
|
+
* Thrown when a trail replace cannot be expressed in the resting order's unit
|
|
10128
|
+
* without guessing. No replace is sent; the resting stop keeps protecting.
|
|
10129
|
+
*/
|
|
10130
|
+
class TrailUnitConversionRefusedError extends AdapticUtilsError {
|
|
10131
|
+
/** The order the replace targeted. */
|
|
10132
|
+
orderId;
|
|
10133
|
+
/** Why the replace was refused. */
|
|
10134
|
+
reason;
|
|
10135
|
+
/** The converted percent, or `null` when no conversion was computed. */
|
|
10136
|
+
pct;
|
|
10137
|
+
/** The conversion reference price, or `null` when none was usable. */
|
|
10138
|
+
ref;
|
|
10139
|
+
constructor(params) {
|
|
10140
|
+
super(`Trailing stop replace refused for ${params.orderId} (${params.reason}): ${params.detail}`, "TRAIL_UNIT_REFUSED", "alpaca", false);
|
|
10141
|
+
this.orderId = params.orderId;
|
|
10142
|
+
this.reason = params.reason;
|
|
10143
|
+
this.pct = params.pct;
|
|
10144
|
+
this.ref = params.ref;
|
|
10145
|
+
}
|
|
10146
|
+
}
|
|
10147
|
+
/** Parse a broker decimal string; `null` when absent, non-finite or not positive. */
|
|
10148
|
+
function positiveOrNull(value) {
|
|
10149
|
+
if (value === null || value === undefined || value === "") {
|
|
10150
|
+
return null;
|
|
10151
|
+
}
|
|
10152
|
+
const parsed = Number(value);
|
|
10153
|
+
return Number.isFinite(parsed) && parsed > 0 ? parsed : null;
|
|
10154
|
+
}
|
|
10155
|
+
/**
|
|
10156
|
+
* Read the unit a resting trailing-stop order trails in.
|
|
10157
|
+
*
|
|
10158
|
+
* @param order - The resting order as returned by the broker.
|
|
10159
|
+
* @returns `"price"` for a dollar trail, `"percent"` for a percent trail, or
|
|
10160
|
+
* `null` when neither (or both) unit fields carry a positive value.
|
|
10161
|
+
*/
|
|
10162
|
+
function readTrailUnit(order) {
|
|
10163
|
+
const hasPrice = positiveOrNull(order.trail_price) !== null;
|
|
10164
|
+
const hasPercent = positiveOrNull(order.trail_percent) !== null;
|
|
10165
|
+
if (hasPrice === hasPercent) {
|
|
10166
|
+
return null;
|
|
10167
|
+
}
|
|
10168
|
+
return hasPrice ? "price" : "percent";
|
|
10169
|
+
}
|
|
10170
|
+
/**
|
|
10171
|
+
* Resolve the replace `trail` value for a requested trail against the resting
|
|
10172
|
+
* order's unit.
|
|
10173
|
+
*
|
|
10174
|
+
* - A dollar distance on a dollar order is sent unchanged.
|
|
10175
|
+
* - A dollar distance on a percent order is converted against
|
|
10176
|
+
* `ref = max(hwm, stop_price)`. For a long the HWM is at or above the live
|
|
10177
|
+
* price; for a short the stop is above it. So `ref` is at or above live on
|
|
10178
|
+
* both sides, and `distance / ref` is at or below `distance / live`: the
|
|
10179
|
+
* resulting stop is at or tighter than `live ∓ distance`. The percent is
|
|
10180
|
+
* rounded down to hundredths, which only tightens it further.
|
|
10181
|
+
* - A percent on a percent order is sent unchanged.
|
|
10182
|
+
* - A percent on a dollar order is refused. There is no conversion that keeps
|
|
10183
|
+
* the caller's intent without a live price this seam does not own.
|
|
10184
|
+
*
|
|
10185
|
+
* @param orderId - The order being replaced (for the refusal record).
|
|
10186
|
+
* @param order - The resting order as returned by the broker.
|
|
10187
|
+
* @param requested - Exactly one of a positive dollar distance or percent.
|
|
10188
|
+
* @returns The resolved `trail` value and its unit.
|
|
10189
|
+
* @throws {TrailUnitConversionRefusedError} When the unit is unknown, no
|
|
10190
|
+
* finite reference exists, the converted percent falls outside
|
|
10191
|
+
* [{@link MIN_CONVERTED_TRAIL_PERCENT}, {@link ALPACA_MAX_TRAIL_PERCENT}],
|
|
10192
|
+
* or a percent targets a dollar order.
|
|
10193
|
+
*/
|
|
10194
|
+
function resolveReplaceTrail(orderId, order, requested) {
|
|
10195
|
+
const unit = readTrailUnit(order);
|
|
10196
|
+
if (unit === null) {
|
|
10197
|
+
throw new TrailUnitConversionRefusedError({
|
|
10198
|
+
orderId,
|
|
10199
|
+
reason: "unit_unknown",
|
|
10200
|
+
pct: null,
|
|
10201
|
+
ref: null,
|
|
10202
|
+
detail: `order carries trail_price=${String(order.trail_price)} trail_percent=${String(order.trail_percent)}; the replace unit cannot be determined`,
|
|
10203
|
+
});
|
|
10204
|
+
}
|
|
10205
|
+
if ("trailPercent" in requested) {
|
|
10206
|
+
if (unit === "price") {
|
|
10207
|
+
throw new TrailUnitConversionRefusedError({
|
|
10208
|
+
orderId,
|
|
10209
|
+
reason: "percent_on_price_order",
|
|
10210
|
+
pct: requested.trailPercent,
|
|
10211
|
+
ref: null,
|
|
10212
|
+
detail: `a ${requested.trailPercent}% trail sent to a dollar-trail order would be stored as $${requested.trailPercent}`,
|
|
10213
|
+
});
|
|
10214
|
+
}
|
|
10215
|
+
return { trail: requested.trailPercent.toString(), unit, referencePrice: null };
|
|
10216
|
+
}
|
|
10217
|
+
if (unit === "price") {
|
|
10218
|
+
return { trail: requested.trailPrice.toString(), unit, referencePrice: null };
|
|
10219
|
+
}
|
|
10220
|
+
const candidates = [positiveOrNull(order.hwm), positiveOrNull(order.stop_price)].filter((value) => value !== null);
|
|
10221
|
+
if (candidates.length === 0) {
|
|
10222
|
+
throw new TrailUnitConversionRefusedError({
|
|
10223
|
+
orderId,
|
|
10224
|
+
reason: "reference_unavailable",
|
|
10225
|
+
pct: null,
|
|
10226
|
+
ref: null,
|
|
10227
|
+
detail: `percent-trail order has no finite hwm (${String(order.hwm)}) or stop_price (${String(order.stop_price)}) to convert $${requested.trailPrice} against`,
|
|
10228
|
+
});
|
|
10229
|
+
}
|
|
10230
|
+
const ref = Math.max(...candidates);
|
|
10231
|
+
const pct = Math.floor((requested.trailPrice / ref) * PERCENT_PER_UNIT * PERCENT_DECIMALS_SCALE) /
|
|
10232
|
+
PERCENT_DECIMALS_SCALE;
|
|
10233
|
+
if (!Number.isFinite(pct) ||
|
|
10234
|
+
pct < MIN_CONVERTED_TRAIL_PERCENT ||
|
|
10235
|
+
pct > ALPACA_MAX_TRAIL_PERCENT) {
|
|
10236
|
+
throw new TrailUnitConversionRefusedError({
|
|
10237
|
+
orderId,
|
|
10238
|
+
reason: "converted_percent_out_of_range",
|
|
10239
|
+
pct,
|
|
10240
|
+
ref,
|
|
10241
|
+
detail: `$${requested.trailPrice} against ref ${ref} is ${pct}%, outside [${MIN_CONVERTED_TRAIL_PERCENT}, ${ALPACA_MAX_TRAIL_PERCENT}]`,
|
|
10242
|
+
});
|
|
10243
|
+
}
|
|
10244
|
+
return { trail: pct.toFixed(2), unit, referencePrice: ref };
|
|
10245
|
+
}
|
|
10246
|
+
|
|
10093
10247
|
const limitPriceSlippagePercent100 = 0.1; // 0.1%
|
|
10094
10248
|
/**
|
|
10095
10249
|
* Alpaca's maximum page size for GET /orders — also our explicit default.
|
|
@@ -11039,12 +11193,18 @@ class AlpacaTradingAPI {
|
|
|
11039
11193
|
return null;
|
|
11040
11194
|
}
|
|
11041
11195
|
const originalOrderId = trailingStopOrder.id;
|
|
11196
|
+
// Alpaca reads the replace `trail` in the resting order's unit. A percent
|
|
11197
|
+
// sent to a dollar-trail order would be stored as dollars, so it is refused
|
|
11198
|
+
// (typed, no replace sent) and the resting dollar trail keeps protecting.
|
|
11199
|
+
const resolvedTrail = resolveReplaceTrail(originalOrderId, trailingStopOrder, {
|
|
11200
|
+
trailPercent: trailPercent100,
|
|
11201
|
+
});
|
|
11042
11202
|
this.log(`Updating trailing stop for ${symbol} from ${currentTrailPercent}% to ${trailPercent100}% (orderId=${originalOrderId})`, {
|
|
11043
11203
|
symbol,
|
|
11044
11204
|
});
|
|
11045
11205
|
try {
|
|
11046
11206
|
const updatedOrder = await this.makeRequest(`/orders/${trailingStopOrder.id}`, "PATCH", {
|
|
11047
|
-
trail:
|
|
11207
|
+
trail: resolvedTrail.trail,
|
|
11048
11208
|
});
|
|
11049
11209
|
// Log the replacement: Alpaca replaces orders on PATCH, so new ID is returned
|
|
11050
11210
|
this.log(`Trailing stop updated for ${symbol}: newOrderId=${updatedOrder.id}, replaces=${updatedOrder.replaces || originalOrderId}`, { symbol });
|
|
@@ -63460,13 +63620,6 @@ var orderUtils$1 = /*#__PURE__*/Object.freeze({
|
|
|
63460
63620
|
});
|
|
63461
63621
|
|
|
63462
63622
|
const LOG_SOURCE$7 = "TrailingStops";
|
|
63463
|
-
/**
|
|
63464
|
-
* Alpaca's hard upper limit for `trail_percent` on trailing-stop orders.
|
|
63465
|
-
* Submissions exceeding this value are rejected with HTTP 422 / code 42210000
|
|
63466
|
-
* ("trail_percent must be <= 25"). See:
|
|
63467
|
-
* https://docs.alpaca.markets/reference/postorder
|
|
63468
|
-
*/
|
|
63469
|
-
const ALPACA_MAX_TRAIL_PERCENT = 25;
|
|
63470
63623
|
/**
|
|
63471
63624
|
* Internal logging helper with consistent source
|
|
63472
63625
|
*/
|
|
@@ -63595,23 +63748,42 @@ async function createTrailingStop(client, params) {
|
|
|
63595
63748
|
}
|
|
63596
63749
|
}
|
|
63597
63750
|
/**
|
|
63598
|
-
* Update an existing trailing stop order
|
|
63751
|
+
* Update the trail distance of an existing trailing stop order.
|
|
63752
|
+
*
|
|
63753
|
+
* ## Unit contract
|
|
63754
|
+
*
|
|
63755
|
+
* Alpaca's replace takes a single unitless `trail` field and reads it in the
|
|
63756
|
+
* unit of the ORIGINAL order; a replace cannot change the unit. This function
|
|
63757
|
+
* therefore reads the resting order first and resolves the request against its
|
|
63758
|
+
* unit ({@link resolveReplaceTrail}):
|
|
63599
63759
|
*
|
|
63600
|
-
*
|
|
63601
|
-
*
|
|
63760
|
+
* - `trailPrice` on a dollar-trail order is sent as dollars, unchanged.
|
|
63761
|
+
* - `trailPrice` on a percent-trail order is converted to a percent against
|
|
63762
|
+
* `max(hwm, stop_price)`, which is at or above the live price on both sides,
|
|
63763
|
+
* so the resulting stop is at or tighter than `live ∓ trailPrice`. The
|
|
63764
|
+
* percent is rounded down to hundredths (tighter).
|
|
63765
|
+
* - `trailPercent` on a percent-trail order is sent unchanged.
|
|
63766
|
+
* - `trailPercent` on a dollar-trail order is refused.
|
|
63767
|
+
*
|
|
63768
|
+
* A refusal throws {@link TrailUnitConversionRefusedError} and sends no
|
|
63769
|
+
* replace, so the resting stop keeps protecting. The unit is never guessed and
|
|
63770
|
+
* the conversion reference is never defaulted.
|
|
63602
63771
|
*
|
|
63603
63772
|
* @param client - AlpacaClient instance
|
|
63604
63773
|
* @param orderId - The ID of the order to update
|
|
63605
|
-
* @param updates - New trail parameters (specify one of trailPercent or trailPrice)
|
|
63606
|
-
* @returns The
|
|
63607
|
-
* @throws {Error} If no update parameters
|
|
63774
|
+
* @param updates - New trail parameters (specify exactly one of trailPercent or trailPrice)
|
|
63775
|
+
* @returns The replacement order
|
|
63776
|
+
* @throws {Error} If no/both update parameters are given, or a value is not positive
|
|
63777
|
+
* @throws {TrailUnitConversionRefusedError} If the request cannot be expressed
|
|
63778
|
+
* in the resting order's unit
|
|
63779
|
+
* @throws {AlpacaApiError} If the order read or the replace fails at the broker
|
|
63608
63780
|
*
|
|
63609
63781
|
* @example
|
|
63610
63782
|
* ```typescript
|
|
63611
|
-
* // Tighten trailing stop to 1.5%
|
|
63783
|
+
* // Tighten a percent trailing stop to 1.5%
|
|
63612
63784
|
* await updateTrailingStop(client, 'order-id-123', { trailPercent: 1.5 });
|
|
63613
63785
|
*
|
|
63614
|
-
* //
|
|
63786
|
+
* // Pin a $3 trail distance (converted when the order trails in percent)
|
|
63615
63787
|
* await updateTrailingStop(client, 'order-id-123', { trailPrice: 3.00 });
|
|
63616
63788
|
* ```
|
|
63617
63789
|
*/
|
|
@@ -63631,21 +63803,41 @@ async function updateTrailingStop(client, orderId, updates) {
|
|
|
63631
63803
|
throw new Error("trailPrice must be greater than 0");
|
|
63632
63804
|
}
|
|
63633
63805
|
const sdk = client.getSDK();
|
|
63634
|
-
const
|
|
63635
|
-
?
|
|
63636
|
-
:
|
|
63806
|
+
const requested = updates.trailPercent !== undefined
|
|
63807
|
+
? { trailPercent: updates.trailPercent }
|
|
63808
|
+
: { trailPrice: updates.trailPrice };
|
|
63809
|
+
const updateDescription = "trailPercent" in requested
|
|
63810
|
+
? `${requested.trailPercent}%`
|
|
63811
|
+
: `$${requested.trailPrice.toFixed(2)}`;
|
|
63637
63812
|
log$g(`Updating trailing stop ${orderId} to trail: ${updateDescription}`, {
|
|
63638
63813
|
type: "info",
|
|
63639
63814
|
});
|
|
63815
|
+
let resting;
|
|
63640
63816
|
try {
|
|
63641
|
-
|
|
63642
|
-
|
|
63643
|
-
|
|
63644
|
-
|
|
63645
|
-
}
|
|
63646
|
-
|
|
63647
|
-
|
|
63648
|
-
}
|
|
63817
|
+
resting = (await sdk.getOrder(orderId));
|
|
63818
|
+
}
|
|
63819
|
+
catch (error) {
|
|
63820
|
+
const err = error;
|
|
63821
|
+
log$g(`Trailing stop update aborted for ${orderId}: order read failed: ${err.message}`, {
|
|
63822
|
+
type: "error",
|
|
63823
|
+
});
|
|
63824
|
+
throw enrichAlpacaError(new Error(`Failed to read trailing stop ${orderId} before update: ${err.message}`), error);
|
|
63825
|
+
}
|
|
63826
|
+
let resolved;
|
|
63827
|
+
try {
|
|
63828
|
+
resolved = resolveReplaceTrail(orderId, resting, requested);
|
|
63829
|
+
}
|
|
63830
|
+
catch (refusal) {
|
|
63831
|
+
log$g(`Trailing stop update refused for ${orderId}: ${refusal.message}`, {
|
|
63832
|
+
type: "warn",
|
|
63833
|
+
});
|
|
63834
|
+
throw refusal;
|
|
63835
|
+
}
|
|
63836
|
+
if (resolved.referencePrice !== null) {
|
|
63837
|
+
log$g(`Trailing stop ${orderId}: ${updateDescription} on percent order → ${resolved.trail}% (ref ${resolved.referencePrice})`, { type: "info" });
|
|
63838
|
+
}
|
|
63839
|
+
try {
|
|
63840
|
+
const replaceParams = { trail: resolved.trail };
|
|
63649
63841
|
const order = await sdk.replaceOrder(orderId, replaceParams);
|
|
63650
63842
|
log$g(`Trailing stop updated: orderId=${order.id}, new replacement created`, {
|
|
63651
63843
|
type: "info",
|
|
@@ -72350,6 +72542,192 @@ const alpaca = {
|
|
|
72350
72542
|
streams: streams$1,
|
|
72351
72543
|
};
|
|
72352
72544
|
|
|
72545
|
+
/**
|
|
72546
|
+
* Rolling healthy-latency evidence per (provider, model), split by prompt size.
|
|
72547
|
+
*
|
|
72548
|
+
* The chain's timeouts and hedge points are only as good as its idea of how
|
|
72549
|
+
* long a healthy answer takes. A flat route budget encodes no such idea: it
|
|
72550
|
+
* treats a thirty-second wait on a model whose healthy answers arrive in four
|
|
72551
|
+
* the same as a thirty-second wait on one that needs twenty-five. This tracker
|
|
72552
|
+
* supplies the measured alternative.
|
|
72553
|
+
*
|
|
72554
|
+
* It is keyed by provider and model rather than by alias and role, because
|
|
72555
|
+
* health is a property of the model at its host: the same model reached as one
|
|
72556
|
+
* alias's primary and another alias's secondary is one population, and
|
|
72557
|
+
* splitting it would halve the evidence each side sees. Samples are split by
|
|
72558
|
+
* prompt size, because generation time grows with input and a model that
|
|
72559
|
+
* struggles on large prompts would otherwise have its large-prompt tail hidden
|
|
72560
|
+
* by a crowd of small, fast calls.
|
|
72561
|
+
*
|
|
72562
|
+
* Only healthy (answered) attempts are recorded. A timeout says the answer
|
|
72563
|
+
* took at least the budget, not how long it took, and folding it in would pull
|
|
72564
|
+
* the quantiles toward whatever budget happened to be configured.
|
|
72565
|
+
*
|
|
72566
|
+
* Unknown stays unknown: a cell with fewer than the minimum samples answers
|
|
72567
|
+
* `null`, and every consumer treats `null` as "no evidence", never as a value.
|
|
72568
|
+
*
|
|
72569
|
+
* @module llm/leg-latency-tracker
|
|
72570
|
+
*/
|
|
72571
|
+
/** Characters per token used to size a prompt before the provider has counted it. */
|
|
72572
|
+
const CHARS_PER_TOKEN = 4;
|
|
72573
|
+
/** The bucket used when a prompt's size could not be estimated. */
|
|
72574
|
+
const UNKNOWN_BUCKET = "unknown";
|
|
72575
|
+
/**
|
|
72576
|
+
* Estimate a prompt's size in tokens from its serialised length.
|
|
72577
|
+
*
|
|
72578
|
+
* Only used to choose a size bucket, and the same estimate is applied when a
|
|
72579
|
+
* sample is recorded and when it is looked up, so its bias cancels. Content
|
|
72580
|
+
* that cannot be serialised yields `null` rather than a guessed size.
|
|
72581
|
+
*
|
|
72582
|
+
* @param parts The prompt, developer instruction and prior turns.
|
|
72583
|
+
* @returns The estimated token count, or null.
|
|
72584
|
+
*/
|
|
72585
|
+
function estimatePromptTokens(parts) {
|
|
72586
|
+
let characters = 0;
|
|
72587
|
+
for (const part of parts) {
|
|
72588
|
+
if (part === undefined) {
|
|
72589
|
+
continue;
|
|
72590
|
+
}
|
|
72591
|
+
if (typeof part === "string") {
|
|
72592
|
+
characters += part.length;
|
|
72593
|
+
continue;
|
|
72594
|
+
}
|
|
72595
|
+
try {
|
|
72596
|
+
const serialised = JSON.stringify(part);
|
|
72597
|
+
if (typeof serialised !== "string") {
|
|
72598
|
+
return null;
|
|
72599
|
+
}
|
|
72600
|
+
characters += serialised.length;
|
|
72601
|
+
}
|
|
72602
|
+
catch {
|
|
72603
|
+
// A circular or otherwise unserialisable part has no knowable size; the
|
|
72604
|
+
// caller files its latency under the unknown bucket rather than a guess.
|
|
72605
|
+
return null;
|
|
72606
|
+
}
|
|
72607
|
+
}
|
|
72608
|
+
return Math.ceil(characters / CHARS_PER_TOKEN);
|
|
72609
|
+
}
|
|
72610
|
+
/**
|
|
72611
|
+
* The nearest-rank quantile of a sorted sample.
|
|
72612
|
+
*
|
|
72613
|
+
* @param sorted Ascending values; must be non-empty.
|
|
72614
|
+
* @param q The quantile, in (0, 1).
|
|
72615
|
+
* @returns The value at that rank.
|
|
72616
|
+
*/
|
|
72617
|
+
function nearestRank(sorted, q) {
|
|
72618
|
+
const rank = Math.min(sorted.length, Math.max(1, Math.ceil(q * sorted.length)));
|
|
72619
|
+
return sorted[rank - 1];
|
|
72620
|
+
}
|
|
72621
|
+
/**
|
|
72622
|
+
* Per-process healthy-latency windows.
|
|
72623
|
+
*/
|
|
72624
|
+
class LegLatencyTracker {
|
|
72625
|
+
cells = new Map();
|
|
72626
|
+
config;
|
|
72627
|
+
now;
|
|
72628
|
+
/**
|
|
72629
|
+
* @param config Window sizing and prompt-size buckets.
|
|
72630
|
+
* @param now Clock, injected so ageing is testable without waiting.
|
|
72631
|
+
*/
|
|
72632
|
+
constructor(config, now = Date.now) {
|
|
72633
|
+
this.config = config;
|
|
72634
|
+
this.now = now;
|
|
72635
|
+
}
|
|
72636
|
+
/**
|
|
72637
|
+
* The size bucket a prompt falls in.
|
|
72638
|
+
*
|
|
72639
|
+
* @param promptTokens Estimated prompt tokens, or null when unknown.
|
|
72640
|
+
* @returns The bucket label.
|
|
72641
|
+
*/
|
|
72642
|
+
bucketOf(promptTokens) {
|
|
72643
|
+
if (promptTokens === null || !Number.isFinite(promptTokens)) {
|
|
72644
|
+
return UNKNOWN_BUCKET;
|
|
72645
|
+
}
|
|
72646
|
+
const index = this.config.promptTokenBuckets.findIndex((edge) => promptTokens < edge);
|
|
72647
|
+
return String(index === -1 ? this.config.promptTokenBuckets.length : index);
|
|
72648
|
+
}
|
|
72649
|
+
/**
|
|
72650
|
+
* Record one healthy (answered) attempt.
|
|
72651
|
+
*
|
|
72652
|
+
* @param provider The provider that answered.
|
|
72653
|
+
* @param modelId The model it answered with.
|
|
72654
|
+
* @param promptTokens Estimated prompt tokens, or null.
|
|
72655
|
+
* @param durationMs How long the answer took.
|
|
72656
|
+
* @returns void
|
|
72657
|
+
*/
|
|
72658
|
+
record(provider, modelId, promptTokens, durationMs) {
|
|
72659
|
+
if (!Number.isFinite(durationMs) || durationMs < 0) {
|
|
72660
|
+
return;
|
|
72661
|
+
}
|
|
72662
|
+
const key = this.cellKey(provider, modelId, promptTokens);
|
|
72663
|
+
const samples = this.fresh(key);
|
|
72664
|
+
samples.push({ atMs: this.now(), durationMs });
|
|
72665
|
+
while (samples.length > this.config.windowSize) {
|
|
72666
|
+
samples.shift();
|
|
72667
|
+
}
|
|
72668
|
+
this.cells.set(key, samples);
|
|
72669
|
+
}
|
|
72670
|
+
/**
|
|
72671
|
+
* A healthy-latency quantile, or null without enough evidence.
|
|
72672
|
+
*
|
|
72673
|
+
* @param provider The provider.
|
|
72674
|
+
* @param modelId The model.
|
|
72675
|
+
* @param promptTokens Estimated prompt tokens, or null.
|
|
72676
|
+
* @param q The quantile, in (0, 1).
|
|
72677
|
+
* @returns The quantile in milliseconds, or null.
|
|
72678
|
+
*/
|
|
72679
|
+
quantile(provider, modelId, promptTokens, q) {
|
|
72680
|
+
const samples = this.fresh(this.cellKey(provider, modelId, promptTokens));
|
|
72681
|
+
if (samples.length < this.config.minSamples) {
|
|
72682
|
+
return null;
|
|
72683
|
+
}
|
|
72684
|
+
const sorted = samples.map((sample) => sample.durationMs).sort((a, b) => a - b);
|
|
72685
|
+
return nearestRank(sorted, q);
|
|
72686
|
+
}
|
|
72687
|
+
/**
|
|
72688
|
+
* How many fresh samples a cell holds.
|
|
72689
|
+
*
|
|
72690
|
+
* @param provider The provider.
|
|
72691
|
+
* @param modelId The model.
|
|
72692
|
+
* @param promptTokens Estimated prompt tokens, or null.
|
|
72693
|
+
* @returns The count.
|
|
72694
|
+
*/
|
|
72695
|
+
sampleCount(provider, modelId, promptTokens) {
|
|
72696
|
+
return this.fresh(this.cellKey(provider, modelId, promptTokens)).length;
|
|
72697
|
+
}
|
|
72698
|
+
/**
|
|
72699
|
+
* Discard all evidence.
|
|
72700
|
+
*
|
|
72701
|
+
* @returns void
|
|
72702
|
+
*/
|
|
72703
|
+
reset() {
|
|
72704
|
+
this.cells.clear();
|
|
72705
|
+
}
|
|
72706
|
+
/**
|
|
72707
|
+
* @param provider The provider.
|
|
72708
|
+
* @param modelId The model.
|
|
72709
|
+
* @param promptTokens Estimated prompt tokens, or null.
|
|
72710
|
+
* @returns The cell key.
|
|
72711
|
+
*/
|
|
72712
|
+
cellKey(provider, modelId, promptTokens) {
|
|
72713
|
+
return `${provider}/${modelId}@${this.bucketOf(promptTokens)}`;
|
|
72714
|
+
}
|
|
72715
|
+
/**
|
|
72716
|
+
* A cell's samples with the stale ones removed.
|
|
72717
|
+
*
|
|
72718
|
+
* @param key The cell key.
|
|
72719
|
+
* @returns The fresh samples (the stored array, pruned in place).
|
|
72720
|
+
*/
|
|
72721
|
+
fresh(key) {
|
|
72722
|
+
const samples = this.cells.get(key) ?? [];
|
|
72723
|
+
const oldest = this.now() - this.config.sampleMaxAgeMs;
|
|
72724
|
+
while (samples.length > 0 && samples[0].atMs < oldest) {
|
|
72725
|
+
samples.shift();
|
|
72726
|
+
}
|
|
72727
|
+
return samples;
|
|
72728
|
+
}
|
|
72729
|
+
}
|
|
72730
|
+
|
|
72353
72731
|
/**
|
|
72354
72732
|
* Per-route circuit breaker for the alias client.
|
|
72355
72733
|
*
|
|
@@ -72379,6 +72757,17 @@ const alpaca = {
|
|
|
72379
72757
|
* for the full `cooldown_ms`. Either way the route then admits a bounded number
|
|
72380
72758
|
* of half-open probes, and one success closes it.
|
|
72381
72759
|
*
|
|
72760
|
+
* A route can also be opened by LATENCY (when `latency_trip` is armed): a
|
|
72761
|
+
* provider whose answers arrive, but later than the latency class's objective
|
|
72762
|
+
* in most recent windows, is failing a hot path without ever producing an
|
|
72763
|
+
* error. A latency-opened breaker is not closed by an answer from an attempt
|
|
72764
|
+
* that started before it opened, because such an answer is exactly the slow
|
|
72765
|
+
* evidence that opened it.
|
|
72766
|
+
*
|
|
72767
|
+
* The half-open probe budget scales with how much traffic the route carried
|
|
72768
|
+
* before it opened (`probe_fraction`), so a route that served forty calls at
|
|
72769
|
+
* once is not re-tested by a single probe whose one slow answer decides it.
|
|
72770
|
+
*
|
|
72382
72771
|
* The clock is injected. Breaker behaviour is entirely about elapsed time, and
|
|
72383
72772
|
* a test that must sleep to observe a cooldown is a test nobody runs.
|
|
72384
72773
|
*
|
|
@@ -72393,6 +72782,8 @@ function freshRecord() {
|
|
|
72393
72782
|
openedAtMs: null,
|
|
72394
72783
|
probesInFlight: 0,
|
|
72395
72784
|
runHasHardFailure: false,
|
|
72785
|
+
openedByLatency: false,
|
|
72786
|
+
concurrencyAtOpen: 0,
|
|
72396
72787
|
};
|
|
72397
72788
|
}
|
|
72398
72789
|
/**
|
|
@@ -72400,6 +72791,11 @@ function freshRecord() {
|
|
|
72400
72791
|
*/
|
|
72401
72792
|
class CircuitBreakerRegistry {
|
|
72402
72793
|
records = new Map();
|
|
72794
|
+
latency = new Map();
|
|
72795
|
+
/** Attempts currently in flight per route, for scaling the probe budget. */
|
|
72796
|
+
inFlight = new Map();
|
|
72797
|
+
/** Peak of {@link inFlight} since the route last opened. */
|
|
72798
|
+
peakInFlight = new Map();
|
|
72403
72799
|
config;
|
|
72404
72800
|
now;
|
|
72405
72801
|
/**
|
|
@@ -72468,7 +72864,20 @@ class CircuitBreakerRegistry {
|
|
|
72468
72864
|
return false;
|
|
72469
72865
|
}
|
|
72470
72866
|
const record = this.recordFor(routeKey);
|
|
72471
|
-
return record.probesInFlight < this.
|
|
72867
|
+
return record.probesInFlight < this.probeBudgetFor(record);
|
|
72868
|
+
}
|
|
72869
|
+
/**
|
|
72870
|
+
* How many half-open probes a route admits at once.
|
|
72871
|
+
*
|
|
72872
|
+
* @param record The route's record.
|
|
72873
|
+
* @returns The larger of the configured floor and the concurrency-scaled budget.
|
|
72874
|
+
*/
|
|
72875
|
+
probeBudgetFor(record) {
|
|
72876
|
+
const fraction = this.config.probe_fraction;
|
|
72877
|
+
if (fraction === undefined || fraction <= 0) {
|
|
72878
|
+
return this.config.half_open_probes;
|
|
72879
|
+
}
|
|
72880
|
+
return Math.max(this.config.half_open_probes, Math.ceil(fraction * record.concurrencyAtOpen));
|
|
72472
72881
|
}
|
|
72473
72882
|
/**
|
|
72474
72883
|
* Register that an attempt is starting, so half-open probes stay bounded.
|
|
@@ -72479,6 +72888,9 @@ class CircuitBreakerRegistry {
|
|
|
72479
72888
|
* {@link onAttemptAbandoned}, or the slot is never returned.
|
|
72480
72889
|
*/
|
|
72481
72890
|
onAttemptStart(routeKey) {
|
|
72891
|
+
const inFlight = (this.inFlight.get(routeKey) ?? 0) + 1;
|
|
72892
|
+
this.inFlight.set(routeKey, inFlight);
|
|
72893
|
+
this.peakInFlight.set(routeKey, Math.max(this.peakInFlight.get(routeKey) ?? 0, inFlight));
|
|
72482
72894
|
if (this.stateOf(routeKey) === "half-open") {
|
|
72483
72895
|
this.recordFor(routeKey).probesInFlight += 1;
|
|
72484
72896
|
return true;
|
|
@@ -72505,6 +72917,17 @@ class CircuitBreakerRegistry {
|
|
|
72505
72917
|
record.probesInFlight -= 1;
|
|
72506
72918
|
}
|
|
72507
72919
|
}
|
|
72920
|
+
/**
|
|
72921
|
+
* Mark an attempt started with {@link onAttemptStart} as finished, whatever
|
|
72922
|
+
* its outcome, so the in-flight count that scales the probe budget stays true.
|
|
72923
|
+
*
|
|
72924
|
+
* @param routeKey The route's stable key.
|
|
72925
|
+
* @returns void
|
|
72926
|
+
*/
|
|
72927
|
+
onAttemptEnd(routeKey) {
|
|
72928
|
+
const inFlight = this.inFlight.get(routeKey) ?? 0;
|
|
72929
|
+
this.inFlight.set(routeKey, Math.max(0, inFlight - 1));
|
|
72930
|
+
}
|
|
72508
72931
|
/**
|
|
72509
72932
|
* Record a success, closing the breaker.
|
|
72510
72933
|
*
|
|
@@ -72513,12 +72936,69 @@ class CircuitBreakerRegistry {
|
|
|
72513
72936
|
* working call answers it; requiring several would keep a recovered provider
|
|
72514
72937
|
* excluded while the chain paid for slower legs.
|
|
72515
72938
|
*
|
|
72939
|
+
* The one exception is a breaker opened by latency: an answer from an
|
|
72940
|
+
* attempt that started before it opened is the slow evidence that opened it,
|
|
72941
|
+
* not evidence of recovery, and is ignored.
|
|
72942
|
+
*
|
|
72516
72943
|
* @param routeKey The route's stable key.
|
|
72944
|
+
* @param startedAtMs When the answering attempt started, on the registry's clock.
|
|
72517
72945
|
* @returns void
|
|
72518
72946
|
*/
|
|
72519
|
-
onSuccess(routeKey) {
|
|
72947
|
+
onSuccess(routeKey, startedAtMs) {
|
|
72948
|
+
const record = this.records.get(routeKey);
|
|
72949
|
+
if (record !== undefined &&
|
|
72950
|
+
record.openedByLatency &&
|
|
72951
|
+
record.openedAtMs !== null &&
|
|
72952
|
+
startedAtMs !== undefined &&
|
|
72953
|
+
startedAtMs < record.openedAtMs) {
|
|
72954
|
+
return;
|
|
72955
|
+
}
|
|
72520
72956
|
this.records.set(routeKey, freshRecord());
|
|
72521
72957
|
}
|
|
72958
|
+
/**
|
|
72959
|
+
* Record how long an attempt that reached the provider took.
|
|
72960
|
+
*
|
|
72961
|
+
* Durations fill fixed-size windows; each full window is judged against the
|
|
72962
|
+
* latency class's objective at the configured quantile, and the breaker opens
|
|
72963
|
+
* when enough of the recent windows were over it. Does nothing unless the
|
|
72964
|
+
* latency trip is armed.
|
|
72965
|
+
*
|
|
72966
|
+
* @param routeKey The route's stable key.
|
|
72967
|
+
* @param durationMs The attempt's duration.
|
|
72968
|
+
* @param latencyClass The alias's latency class, which selects the objective.
|
|
72969
|
+
* @returns void
|
|
72970
|
+
*/
|
|
72971
|
+
onLatencySample(routeKey, durationMs, latencyClass) {
|
|
72972
|
+
const trip = this.config.latency_trip;
|
|
72973
|
+
if (trip === undefined || !trip.enabled || latencyClass === undefined) {
|
|
72974
|
+
return;
|
|
72975
|
+
}
|
|
72976
|
+
if (!Number.isFinite(durationMs) || durationMs < 0) {
|
|
72977
|
+
return;
|
|
72978
|
+
}
|
|
72979
|
+
let state = this.latency.get(routeKey);
|
|
72980
|
+
if (state === undefined) {
|
|
72981
|
+
state = { samples: [], verdicts: [] };
|
|
72982
|
+
this.latency.set(routeKey, state);
|
|
72983
|
+
}
|
|
72984
|
+
state.samples.push(durationMs);
|
|
72985
|
+
if (state.samples.length < trip.window_size) {
|
|
72986
|
+
return;
|
|
72987
|
+
}
|
|
72988
|
+
const sorted = [...state.samples].sort((a, b) => a - b);
|
|
72989
|
+
state.samples = [];
|
|
72990
|
+
state.verdicts.push(nearestRank(sorted, trip.quantile) > trip.slo_ms[latencyClass]);
|
|
72991
|
+
while (state.verdicts.length > trip.of_windows) {
|
|
72992
|
+
state.verdicts.shift();
|
|
72993
|
+
}
|
|
72994
|
+
const over = state.verdicts.filter(Boolean).length;
|
|
72995
|
+
if (over >= trip.trip_windows && this.stateOf(routeKey) === "closed") {
|
|
72996
|
+
const record = this.recordFor(routeKey);
|
|
72997
|
+
this.open(routeKey, record);
|
|
72998
|
+
record.openedByLatency = true;
|
|
72999
|
+
state.verdicts = [];
|
|
73000
|
+
}
|
|
73001
|
+
}
|
|
72522
73002
|
/**
|
|
72523
73003
|
* Record a failure, opening the breaker once the threshold is reached.
|
|
72524
73004
|
*
|
|
@@ -72540,9 +73020,22 @@ class CircuitBreakerRegistry {
|
|
|
72540
73020
|
record.runHasHardFailure = true;
|
|
72541
73021
|
}
|
|
72542
73022
|
if (wasHalfOpen || record.consecutiveFailures >= this.config.failure_threshold) {
|
|
72543
|
-
|
|
73023
|
+
this.open(routeKey, record);
|
|
73024
|
+
record.openedByLatency = false;
|
|
72544
73025
|
}
|
|
72545
73026
|
}
|
|
73027
|
+
/**
|
|
73028
|
+
* Open a route, capturing the concurrency its probe budget scales with.
|
|
73029
|
+
*
|
|
73030
|
+
* @param routeKey The route's stable key.
|
|
73031
|
+
* @param record Its record.
|
|
73032
|
+
* @returns void
|
|
73033
|
+
*/
|
|
73034
|
+
open(routeKey, record) {
|
|
73035
|
+
record.openedAtMs = this.now();
|
|
73036
|
+
record.concurrencyAtOpen = Math.max(record.concurrencyAtOpen, this.peakInFlight.get(routeKey) ?? 0);
|
|
73037
|
+
this.peakInFlight.set(routeKey, this.inFlight.get(routeKey) ?? 0);
|
|
73038
|
+
}
|
|
72546
73039
|
/**
|
|
72547
73040
|
* Inspect a route's breaker.
|
|
72548
73041
|
*
|
|
@@ -72563,6 +73056,8 @@ class CircuitBreakerRegistry {
|
|
|
72563
73056
|
? "hard"
|
|
72564
73057
|
: "capacity",
|
|
72565
73058
|
cooldownMs: this.cooldownFor(record),
|
|
73059
|
+
openedByLatency: record.openedByLatency,
|
|
73060
|
+
probeBudget: this.probeBudgetFor(record),
|
|
72566
73061
|
};
|
|
72567
73062
|
}
|
|
72568
73063
|
/**
|
|
@@ -72582,6 +73077,9 @@ class CircuitBreakerRegistry {
|
|
|
72582
73077
|
*/
|
|
72583
73078
|
reset() {
|
|
72584
73079
|
this.records.clear();
|
|
73080
|
+
this.latency.clear();
|
|
73081
|
+
this.inFlight.clear();
|
|
73082
|
+
this.peakInFlight.clear();
|
|
72585
73083
|
}
|
|
72586
73084
|
/**
|
|
72587
73085
|
* @param routeKey The route's stable key.
|
|
@@ -73291,6 +73789,35 @@ async function withProviderGuards(provider, call, maxWaitMs, scope = {}) {
|
|
|
73291
73789
|
release();
|
|
73292
73790
|
}
|
|
73293
73791
|
}
|
|
73792
|
+
/**
|
|
73793
|
+
* Whether a provider guard has room for a DUPLICATE attempt above a reserve.
|
|
73794
|
+
*
|
|
73795
|
+
* A duplicate is a hedge: a second request for an answer another attempt is
|
|
73796
|
+
* already fetching. It is only worth sending with capacity no first attempt
|
|
73797
|
+
* needs, so it is admitted only when nobody is queued behind either bound and
|
|
73798
|
+
* both the free concurrency (after the duplicate) and the rate tokens stay at
|
|
73799
|
+
* or above the reserved fraction of the guard's ceilings. A provider that is
|
|
73800
|
+
* already busy therefore never sees duplicates, which is when they would do
|
|
73801
|
+
* the most harm.
|
|
73802
|
+
*
|
|
73803
|
+
* @param provider The provider key.
|
|
73804
|
+
* @param modelId The model the duplicate would address.
|
|
73805
|
+
* @param reserveFraction Fraction of each ceiling kept free, in [0, 1).
|
|
73806
|
+
* @returns Whether the duplicate may start.
|
|
73807
|
+
*/
|
|
73808
|
+
function hasDuplicateHeadroom(provider, modelId, reserveFraction) {
|
|
73809
|
+
const identity = guardIdentity(provider, modelId);
|
|
73810
|
+
const limits = limitsFor(provider, identity.modelId);
|
|
73811
|
+
const limiter = rateLimiterFor(identity);
|
|
73812
|
+
const gate = concurrencyGateFor(identity);
|
|
73813
|
+
if (limiter.getQueueLength() > 0 || gate.queueLength() > 0) {
|
|
73814
|
+
return false;
|
|
73815
|
+
}
|
|
73816
|
+
const freeAfter = limits.max_concurrent - gate.inFlightCount() - 1;
|
|
73817
|
+
const tokensAfter = limiter.getAvailableTokens() - TOKENS_PER_REQUEST;
|
|
73818
|
+
return (freeAfter >= Math.ceil(limits.max_concurrent * reserveFraction) &&
|
|
73819
|
+
tokensAfter >= Math.ceil(limits.requests_per_minute * reserveFraction));
|
|
73820
|
+
}
|
|
73294
73821
|
/**
|
|
73295
73822
|
* Inspect the guards currently in use.
|
|
73296
73823
|
*
|
|
@@ -73433,6 +73960,703 @@ function parseStructuredContent(content, responseFormat, usage) {
|
|
|
73433
73960
|
}
|
|
73434
73961
|
}
|
|
73435
73962
|
|
|
73963
|
+
/**
|
|
73964
|
+
* One attempt against one leg: dispatch under a hard timeout, and the
|
|
73965
|
+
* classification of how it ended.
|
|
73966
|
+
*
|
|
73967
|
+
* Split from the chain walker so the same-model hedge runner and the walker
|
|
73968
|
+
* share exactly one definition of what a timeout, a capacity refusal, a
|
|
73969
|
+
* cancellation and a superseded attempt are. Two copies of that
|
|
73970
|
+
* classification would drift, and the breaker would then read the same event
|
|
73971
|
+
* differently depending on which code path produced it.
|
|
73972
|
+
*
|
|
73973
|
+
* @module llm/leg-attempt
|
|
73974
|
+
*/
|
|
73975
|
+
/** Raised internally when an attempt exceeds its hard budget. */
|
|
73976
|
+
class LegTimeoutError extends Error {
|
|
73977
|
+
/**
|
|
73978
|
+
* @param routeKey The leg that timed out.
|
|
73979
|
+
* @param budgetMs Its budget in milliseconds.
|
|
73980
|
+
*/
|
|
73981
|
+
constructor(routeKey, budgetMs) {
|
|
73982
|
+
super(`route ${routeKey} exceeded its ${budgetMs} ms budget`);
|
|
73983
|
+
this.name = "LegTimeoutError";
|
|
73984
|
+
}
|
|
73985
|
+
}
|
|
73986
|
+
/**
|
|
73987
|
+
* Raised internally when an attempt ran past its MEASURED timeout and was
|
|
73988
|
+
* replaced by another attempt on the same model.
|
|
73989
|
+
*
|
|
73990
|
+
* Not a verdict on the provider: the measured timeout is the chain's own
|
|
73991
|
+
* impatience, applied only because a same-model alternative could take over,
|
|
73992
|
+
* so the breaker learns nothing from it.
|
|
73993
|
+
*/
|
|
73994
|
+
class AttemptSupersededError extends Error {
|
|
73995
|
+
/**
|
|
73996
|
+
* @param routeKey The attempt's leg.
|
|
73997
|
+
* @param afterMs How long it ran before it was replaced.
|
|
73998
|
+
*/
|
|
73999
|
+
constructor(routeKey, afterMs) {
|
|
74000
|
+
super(`route ${routeKey} exceeded its measured ${afterMs} ms attempt timeout and was ` +
|
|
74001
|
+
"superseded by a same-model attempt");
|
|
74002
|
+
this.name = "AttemptSupersededError";
|
|
74003
|
+
}
|
|
74004
|
+
}
|
|
74005
|
+
/**
|
|
74006
|
+
* Raised internally on an attempt that was still running when another
|
|
74007
|
+
* attempt on the same model answered first. Not a verdict on the provider.
|
|
74008
|
+
*/
|
|
74009
|
+
class HedgeLoserError extends Error {
|
|
74010
|
+
/**
|
|
74011
|
+
* @param routeKey The losing attempt's leg.
|
|
74012
|
+
*/
|
|
74013
|
+
constructor(routeKey) {
|
|
74014
|
+
super(`route ${routeKey} was cancelled: a same-model attempt answered first`);
|
|
74015
|
+
this.name = "HedgeLoserError";
|
|
74016
|
+
}
|
|
74017
|
+
}
|
|
74018
|
+
/**
|
|
74019
|
+
* Start one attempt under a hard timeout, honouring the caller's own cancellation.
|
|
74020
|
+
*
|
|
74021
|
+
* The timer is always cleared and the abort listener always removed, including
|
|
74022
|
+
* on the success path. A long-lived process that leaked one timer per LLM call
|
|
74023
|
+
* would accumulate them at exactly the rate it does useful work.
|
|
74024
|
+
*
|
|
74025
|
+
* @param leg The leg to run.
|
|
74026
|
+
* @param params Normalised parameters for this leg.
|
|
74027
|
+
* @param request The call's request fields and cancellation.
|
|
74028
|
+
* @param budgetMs The attempt's hard budget.
|
|
74029
|
+
* @returns A handle on the attempt.
|
|
74030
|
+
*/
|
|
74031
|
+
function startAttempt(leg, params, request, budgetMs) {
|
|
74032
|
+
const controller = new AbortController();
|
|
74033
|
+
let chainReason;
|
|
74034
|
+
let hardTimeout = false;
|
|
74035
|
+
const timer = setTimeout(() => {
|
|
74036
|
+
hardTimeout = true;
|
|
74037
|
+
controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
|
|
74038
|
+
}, budgetMs);
|
|
74039
|
+
const forwardAbort = () => {
|
|
74040
|
+
controller.abort(request.callerSignal?.reason);
|
|
74041
|
+
};
|
|
74042
|
+
if (request.callerSignal !== undefined) {
|
|
74043
|
+
if (request.callerSignal.aborted) {
|
|
74044
|
+
forwardAbort();
|
|
74045
|
+
}
|
|
74046
|
+
else {
|
|
74047
|
+
request.callerSignal.addEventListener("abort", forwardAbort, { once: true });
|
|
74048
|
+
}
|
|
74049
|
+
}
|
|
74050
|
+
const run = async () => {
|
|
74051
|
+
try {
|
|
74052
|
+
// The guards wrap the transport rather than the whole attempt, so the
|
|
74053
|
+
// hard timeout above still bounds the total wait: a caller queued behind
|
|
74054
|
+
// the rate limiter is spending its budget just as surely as one waiting
|
|
74055
|
+
// on the provider, and only one clock should govern both. The attempt's
|
|
74056
|
+
// own signal is handed to the guard as well, so an attempt whose budget
|
|
74057
|
+
// or caller is gone leaves the queue at once instead of holding its place.
|
|
74058
|
+
const response = await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
|
|
74059
|
+
route: leg.route,
|
|
74060
|
+
content: request.content,
|
|
74061
|
+
responseFormat: request.responseFormat,
|
|
74062
|
+
params,
|
|
74063
|
+
developerPrompt: request.developerPrompt,
|
|
74064
|
+
context: request.context,
|
|
74065
|
+
signal: controller.signal,
|
|
74066
|
+
correlationId: request.correlationId,
|
|
74067
|
+
}), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
|
|
74068
|
+
assertToolChoiceHonoured(leg.route, params, response);
|
|
74069
|
+
return response;
|
|
74070
|
+
}
|
|
74071
|
+
finally {
|
|
74072
|
+
clearTimeout(timer);
|
|
74073
|
+
request.callerSignal?.removeEventListener("abort", forwardAbort);
|
|
74074
|
+
}
|
|
74075
|
+
};
|
|
74076
|
+
return {
|
|
74077
|
+
promise: run(),
|
|
74078
|
+
abort: (reason) => {
|
|
74079
|
+
if (chainReason === undefined && !controller.signal.aborted) {
|
|
74080
|
+
chainReason = reason;
|
|
74081
|
+
controller.abort(reason);
|
|
74082
|
+
}
|
|
74083
|
+
},
|
|
74084
|
+
abortReason: () => chainReason,
|
|
74085
|
+
timedOut: () => hardTimeout,
|
|
74086
|
+
};
|
|
74087
|
+
}
|
|
74088
|
+
/**
|
|
74089
|
+
* HTTP statuses a provider (or the gateway relaying it) uses to say it is full
|
|
74090
|
+
* rather than that the request or the route is wrong: request timeout, too
|
|
74091
|
+
* early, too many requests, service unavailable, and Anthropic's overloaded.
|
|
74092
|
+
*/
|
|
74093
|
+
const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
|
|
74094
|
+
/**
|
|
74095
|
+
* Wording providers use for a capacity refusal when the status is lost on the
|
|
74096
|
+
* way (a relayed body, a client library's own error). DeepInfra's is
|
|
74097
|
+
* "Model busy, retry later"; Anthropic's is "Overloaded".
|
|
74098
|
+
*/
|
|
74099
|
+
const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
|
|
74100
|
+
/**
|
|
74101
|
+
* Whether a failure is the provider saying it is full rather than broken.
|
|
74102
|
+
*
|
|
74103
|
+
* Read by shape rather than by class, because the same signal reaches the
|
|
74104
|
+
* chain from more than one transport and not every transport's error class is
|
|
74105
|
+
* importable here.
|
|
74106
|
+
*
|
|
74107
|
+
* @param error The thrown value.
|
|
74108
|
+
* @param reason Its message.
|
|
74109
|
+
* @returns Whether it is a capacity signal.
|
|
74110
|
+
*/
|
|
74111
|
+
function isCapacitySignal(error, reason) {
|
|
74112
|
+
if (typeof error === "object" && error !== null) {
|
|
74113
|
+
const status = error.status;
|
|
74114
|
+
if (typeof status === "number" && CAPACITY_STATUSES.has(status)) {
|
|
74115
|
+
return true;
|
|
74116
|
+
}
|
|
74117
|
+
}
|
|
74118
|
+
return CAPACITY_WORDING.test(reason);
|
|
74119
|
+
}
|
|
74120
|
+
/**
|
|
74121
|
+
* Classify why a leg failed.
|
|
74122
|
+
*
|
|
74123
|
+
* The distinction matters to the breaker: a timeout and a 5xx are evidence the
|
|
74124
|
+
* provider is unhealthy, while the caller cancelling is not. Counting a
|
|
74125
|
+
* cancellation as a provider failure would let a burst of user-cancelled
|
|
74126
|
+
* requests open the breaker on a perfectly healthy route.
|
|
74127
|
+
*
|
|
74128
|
+
* Among failures that do count, a capacity signal (the provider said it is
|
|
74129
|
+
* busy, or the leg ran out its budget waiting on it) is told apart from a hard
|
|
74130
|
+
* failure so the breaker can re-admit a busy route sooner than a broken one. A
|
|
74131
|
+
* timeout is read as capacity: on a reachable provider it is what a full queue
|
|
74132
|
+
* looks like from outside, and a provider that is actually down still costs no
|
|
74133
|
+
* more than one probe per capacity cooldown.
|
|
74134
|
+
*
|
|
74135
|
+
* An attempt the chain itself cancelled — replaced after its measured
|
|
74136
|
+
* timeout, or beaten by a same-model attempt — is not a verdict on the
|
|
74137
|
+
* provider either, and is classified by the chain's reason rather than by
|
|
74138
|
+
* whatever the transport happened to throw on the way out.
|
|
74139
|
+
*
|
|
74140
|
+
* @param error The thrown value.
|
|
74141
|
+
* @param callerSignal The caller's cancellation signal, if any.
|
|
74142
|
+
* @returns The outcome and whether it counts against route health.
|
|
74143
|
+
*/
|
|
74144
|
+
function classify(error, callerSignal) {
|
|
74145
|
+
if (callerSignal !== undefined && callerSignal.aborted) {
|
|
74146
|
+
return {
|
|
74147
|
+
outcome: "skipped",
|
|
74148
|
+
reason: "caller cancelled",
|
|
74149
|
+
countsAgainstHealth: false,
|
|
74150
|
+
failureKind: "hard",
|
|
74151
|
+
};
|
|
74152
|
+
}
|
|
74153
|
+
if (error instanceof AttemptSupersededError) {
|
|
74154
|
+
return {
|
|
74155
|
+
outcome: "timeout",
|
|
74156
|
+
reason: error.message,
|
|
74157
|
+
countsAgainstHealth: false,
|
|
74158
|
+
failureKind: "capacity",
|
|
74159
|
+
};
|
|
74160
|
+
}
|
|
74161
|
+
if (error instanceof HedgeLoserError) {
|
|
74162
|
+
return {
|
|
74163
|
+
outcome: "skipped",
|
|
74164
|
+
reason: error.message,
|
|
74165
|
+
countsAgainstHealth: false,
|
|
74166
|
+
failureKind: "capacity",
|
|
74167
|
+
};
|
|
74168
|
+
}
|
|
74169
|
+
if (error instanceof LegTimeoutError) {
|
|
74170
|
+
return {
|
|
74171
|
+
outcome: "timeout",
|
|
74172
|
+
reason: error.message,
|
|
74173
|
+
countsAgainstHealth: true,
|
|
74174
|
+
failureKind: "capacity",
|
|
74175
|
+
};
|
|
74176
|
+
}
|
|
74177
|
+
if (error instanceof UnsupportedCapabilityError) {
|
|
74178
|
+
return {
|
|
74179
|
+
outcome: "skipped",
|
|
74180
|
+
reason: error.message,
|
|
74181
|
+
countsAgainstHealth: false,
|
|
74182
|
+
failureKind: "hard",
|
|
74183
|
+
};
|
|
74184
|
+
}
|
|
74185
|
+
if (error instanceof ToolChoiceIgnoredError) {
|
|
74186
|
+
// The route answered; it broke a declared guarantee rather than failing to
|
|
74187
|
+
// be available, so its breaker is not charged for it.
|
|
74188
|
+
return {
|
|
74189
|
+
outcome: "error",
|
|
74190
|
+
reason: error.message,
|
|
74191
|
+
countsAgainstHealth: false,
|
|
74192
|
+
failureKind: "hard",
|
|
74193
|
+
};
|
|
74194
|
+
}
|
|
74195
|
+
// Self-inflicted pacing, not provider ill-health. Counting it would let the
|
|
74196
|
+
// client's own throttling open a breaker on a perfectly healthy provider and
|
|
74197
|
+
// permanently reroute traffic nobody chose to reroute.
|
|
74198
|
+
if (error instanceof RateGuardTimeoutError) {
|
|
74199
|
+
return {
|
|
74200
|
+
outcome: "skipped",
|
|
74201
|
+
reason: error.message,
|
|
74202
|
+
countsAgainstHealth: false,
|
|
74203
|
+
failureKind: "hard",
|
|
74204
|
+
};
|
|
74205
|
+
}
|
|
74206
|
+
const reason = error instanceof Error ? error.message : String(error);
|
|
74207
|
+
if (/abort/i.test(reason)) {
|
|
74208
|
+
return {
|
|
74209
|
+
outcome: "timeout",
|
|
74210
|
+
reason: `aborted: ${reason}`,
|
|
74211
|
+
countsAgainstHealth: true,
|
|
74212
|
+
failureKind: "capacity",
|
|
74213
|
+
};
|
|
74214
|
+
}
|
|
74215
|
+
if (error instanceof LlmResponseFormatError) {
|
|
74216
|
+
// The provider answered, badly. That is a route defect, not a full queue,
|
|
74217
|
+
// whatever words the unparseable content happens to contain.
|
|
74218
|
+
return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
|
|
74219
|
+
}
|
|
74220
|
+
return {
|
|
74221
|
+
outcome: "error",
|
|
74222
|
+
reason,
|
|
74223
|
+
countsAgainstHealth: true,
|
|
74224
|
+
failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
|
|
74225
|
+
};
|
|
74226
|
+
}
|
|
74227
|
+
/**
|
|
74228
|
+
* Whether the caller has stopped waiting.
|
|
74229
|
+
*
|
|
74230
|
+
* Read through a function rather than inline, because `AbortSignal.aborted` is
|
|
74231
|
+
* a live getter: it can flip to true while a leg is in flight, but a compiler
|
|
74232
|
+
* that narrowed it at the top of the loop would prove the later check
|
|
74233
|
+
* unreachable and invite its removal. The check is not redundant — it is the
|
|
74234
|
+
* only thing that stops the chain spending money on an answer nobody will read.
|
|
74235
|
+
*
|
|
74236
|
+
* @param signal The caller's signal, if any.
|
|
74237
|
+
* @returns Whether the call has been cancelled.
|
|
74238
|
+
*/
|
|
74239
|
+
function isAborted(signal) {
|
|
74240
|
+
return signal !== undefined && signal.aborted;
|
|
74241
|
+
}
|
|
74242
|
+
/**
|
|
74243
|
+
* The usage a failed leg was billed for, when the leg reached an answer.
|
|
74244
|
+
*
|
|
74245
|
+
* A leg that failed after the provider answered — content that does not parse,
|
|
74246
|
+
* or prose where a tool call was mandatory — was still charged. A leg that
|
|
74247
|
+
* never answered (timeout, outage, skip) carries no usage, and none is invented.
|
|
74248
|
+
*
|
|
74249
|
+
* @param error The thrown value.
|
|
74250
|
+
* @returns The billed usage, or undefined when the leg never produced an answer.
|
|
74251
|
+
*/
|
|
74252
|
+
function billedUsageOf(error) {
|
|
74253
|
+
if (error instanceof LlmResponseFormatError || error instanceof ToolChoiceIgnoredError) {
|
|
74254
|
+
return error.usage;
|
|
74255
|
+
}
|
|
74256
|
+
return undefined;
|
|
74257
|
+
}
|
|
74258
|
+
|
|
74259
|
+
/**
|
|
74260
|
+
* Same-model attempts for one leg: hedging, measured attempt timeouts, and
|
|
74261
|
+
* reserving deadline for a same-model alternative.
|
|
74262
|
+
*
|
|
74263
|
+
* A leg is a model. Before the chain gives up on it and reaches a DIFFERENT
|
|
74264
|
+
* model — which changes the answer's quality, not just its latency — it is
|
|
74265
|
+
* worth spending the leg's budget on every way of getting that same model to
|
|
74266
|
+
* answer: the same model at another provider (an "equivalent"), or a second
|
|
74267
|
+
* request to the same provider when that provider has capacity to spare (a
|
|
74268
|
+
* "duplicate"). This module runs those attempts as one group.
|
|
74269
|
+
*
|
|
74270
|
+
* Three mechanics, all bounded by the leg's budget and none of them selecting
|
|
74271
|
+
* a different model:
|
|
74272
|
+
*
|
|
74273
|
+
* - **Hedging.** Once an attempt has run past the model's healthy p90 (from
|
|
74274
|
+
* the latency tracker), the next same-model attempt starts beside it. The
|
|
74275
|
+
* first answer wins and every other attempt is cancelled through its own
|
|
74276
|
+
* abort signal. A hedge that loses, or an attempt cancelled because another
|
|
74277
|
+
* won, says nothing about the provider and never touches its breaker.
|
|
74278
|
+
*
|
|
74279
|
+
* - **Measured attempt timeout.** With latency evidence, an attempt that has
|
|
74280
|
+
* run past `k × p99` of healthy latency is replaced by a same-model
|
|
74281
|
+
* alternative instead of holding the budget to its end. It is replaced only
|
|
74282
|
+
* when an alternative exists: with none, cutting it short would only move
|
|
74283
|
+
* the call to a different model sooner, and it runs to the leg budget as
|
|
74284
|
+
* before.
|
|
74285
|
+
*
|
|
74286
|
+
* - **Deadline reservation.** When a same-model equivalent exists, no single
|
|
74287
|
+
* attempt holds more than `max_attempt_share` of the remaining budget before
|
|
74288
|
+
* the equivalent starts beside it, so a slow first attempt cannot consume the
|
|
74289
|
+
* whole deadline while the equivalent that could have answered never runs.
|
|
74290
|
+
*
|
|
74291
|
+
* With no latency evidence and no equivalent, none of the three can act and
|
|
74292
|
+
* the group is exactly one attempt with the leg's full budget: the serial
|
|
74293
|
+
* chain's behaviour, which a fresh process therefore starts from.
|
|
74294
|
+
*
|
|
74295
|
+
* @module llm/hedge
|
|
74296
|
+
*/
|
|
74297
|
+
/**
|
|
74298
|
+
* Read the policy from the table's defaults.
|
|
74299
|
+
*
|
|
74300
|
+
* @param defaults The `hedging` defaults.
|
|
74301
|
+
* @returns The policy.
|
|
74302
|
+
*/
|
|
74303
|
+
function sameModelPolicyFrom(defaults) {
|
|
74304
|
+
return {
|
|
74305
|
+
maxExtraAttempts: defaults.max_same_model_hedges,
|
|
74306
|
+
hedgeQuantile: defaults.hedge_quantile,
|
|
74307
|
+
timeoutQuantile: defaults.timeout_quantile,
|
|
74308
|
+
kTimeout: defaults.k_timeout,
|
|
74309
|
+
attemptTimeoutFloorMs: defaults.attempt_timeout_floor_ms,
|
|
74310
|
+
maxAttemptShare: defaults.max_attempt_share,
|
|
74311
|
+
duplicateReserve: defaults.duplicate_headroom_reserve,
|
|
74312
|
+
};
|
|
74313
|
+
}
|
|
74314
|
+
/**
|
|
74315
|
+
* The parameters of a prepared leg, when it can serve.
|
|
74316
|
+
*
|
|
74317
|
+
* @param leg The leg.
|
|
74318
|
+
* @returns Its parameters, or undefined when it cannot serve this request.
|
|
74319
|
+
*/
|
|
74320
|
+
function paramsOf(leg) {
|
|
74321
|
+
return leg.params instanceof UnsupportedCapabilityError ? undefined : leg.params;
|
|
74322
|
+
}
|
|
74323
|
+
/**
|
|
74324
|
+
* Run every same-model attempt for one leg until one answers or the budget,
|
|
74325
|
+
* the alternatives, or the caller run out.
|
|
74326
|
+
*
|
|
74327
|
+
* The caller has already found the leg servable (breaker allows it, budget
|
|
74328
|
+
* positive). The returned promise settles only once every attempt it started
|
|
74329
|
+
* has been recorded, so the call's attempt record is complete when it returns.
|
|
74330
|
+
*
|
|
74331
|
+
* @param leg The leg, with its equivalents.
|
|
74332
|
+
* @param groupBudgetMs The leg's budget: its route budget cut to the deadline.
|
|
74333
|
+
* @param budgetIsDeadline Whether that budget was cut by the caller's deadline.
|
|
74334
|
+
* @param ctx The chain around the group.
|
|
74335
|
+
* @returns How the group ended.
|
|
74336
|
+
*/
|
|
74337
|
+
function runSameModelGroup(leg, groupBudgetMs, budgetIsDeadline, ctx) {
|
|
74338
|
+
const { breakers, now, policy, tracker, promptTokens, request } = ctx;
|
|
74339
|
+
const endsAt = now() + groupBudgetMs;
|
|
74340
|
+
const equivalents = (leg.equivalents ?? []).filter((equivalent) => paramsOf(equivalent) !== undefined);
|
|
74341
|
+
const live = new Set();
|
|
74342
|
+
const billed = [];
|
|
74343
|
+
const timeoutCharged = new Set();
|
|
74344
|
+
let nextEquivalent = 0;
|
|
74345
|
+
let extraLaunched = 0;
|
|
74346
|
+
let settled = false;
|
|
74347
|
+
let finished = false;
|
|
74348
|
+
let deadlineBound = false;
|
|
74349
|
+
let hedgeTimer;
|
|
74350
|
+
let answer;
|
|
74351
|
+
return new Promise((resolve) => {
|
|
74352
|
+
/**
|
|
74353
|
+
* A healthy-latency quantile for a leg's model, when there is evidence.
|
|
74354
|
+
*
|
|
74355
|
+
* @param route The leg's route.
|
|
74356
|
+
* @param q The quantile.
|
|
74357
|
+
* @returns Milliseconds, or null.
|
|
74358
|
+
*/
|
|
74359
|
+
const quantileOf = (route, q) => tracker === undefined ? null : tracker.quantile(route.providerName, route.modelId, promptTokens, q);
|
|
74360
|
+
/**
|
|
74361
|
+
* The next same-model attempt that could start now: an equivalent first,
|
|
74362
|
+
* then a duplicate, which needs latency evidence and spare provider capacity.
|
|
74363
|
+
*
|
|
74364
|
+
* @returns The candidate, or undefined.
|
|
74365
|
+
*/
|
|
74366
|
+
const peek = () => {
|
|
74367
|
+
if (policy === undefined || extraLaunched >= policy.maxExtraAttempts) {
|
|
74368
|
+
return undefined;
|
|
74369
|
+
}
|
|
74370
|
+
for (let index = nextEquivalent; index < equivalents.length; index += 1) {
|
|
74371
|
+
const equivalent = equivalents[index];
|
|
74372
|
+
if (breakers.allows(equivalent.route.routeKey)) {
|
|
74373
|
+
return { leg: equivalent, kind: "equivalent", index };
|
|
74374
|
+
}
|
|
74375
|
+
}
|
|
74376
|
+
if (quantileOf(leg.route, policy.hedgeQuantile) !== null &&
|
|
74377
|
+
breakers.allows(leg.route.routeKey) &&
|
|
74378
|
+
ctx.admitDuplicate(leg.route, policy.duplicateReserve)) {
|
|
74379
|
+
return { leg, kind: "duplicate", index: -1 };
|
|
74380
|
+
}
|
|
74381
|
+
return undefined;
|
|
74382
|
+
};
|
|
74383
|
+
/** Resolve once nothing is left in flight. */
|
|
74384
|
+
const finishIfIdle = () => {
|
|
74385
|
+
if (finished || live.size > 0) {
|
|
74386
|
+
return;
|
|
74387
|
+
}
|
|
74388
|
+
finished = true;
|
|
74389
|
+
if (hedgeTimer !== undefined) {
|
|
74390
|
+
clearTimeout(hedgeTimer);
|
|
74391
|
+
}
|
|
74392
|
+
resolve({ answer, billed, deadlineBound });
|
|
74393
|
+
};
|
|
74394
|
+
/**
|
|
74395
|
+
* Arm the hedge for the most recent attempt: at the model's healthy p90,
|
|
74396
|
+
* or — when an equivalent is waiting — no later than the reserved share of
|
|
74397
|
+
* the remaining budget.
|
|
74398
|
+
*
|
|
74399
|
+
* @param latest The attempt just started.
|
|
74400
|
+
*/
|
|
74401
|
+
const scheduleHedge = (latest) => {
|
|
74402
|
+
if (hedgeTimer !== undefined) {
|
|
74403
|
+
clearTimeout(hedgeTimer);
|
|
74404
|
+
hedgeTimer = undefined;
|
|
74405
|
+
}
|
|
74406
|
+
const candidate = peek();
|
|
74407
|
+
if (policy === undefined || candidate === undefined) {
|
|
74408
|
+
return;
|
|
74409
|
+
}
|
|
74410
|
+
const measured = quantileOf(latest.leg.route, policy.hedgeQuantile);
|
|
74411
|
+
const reserved = candidate.kind === "equivalent"
|
|
74412
|
+
? policy.maxAttemptShare * (endsAt - now())
|
|
74413
|
+
: Number.POSITIVE_INFINITY;
|
|
74414
|
+
const delay = Math.min(measured ?? Number.POSITIVE_INFINITY, reserved);
|
|
74415
|
+
if (!Number.isFinite(delay)) {
|
|
74416
|
+
return;
|
|
74417
|
+
}
|
|
74418
|
+
hedgeTimer = setTimeout(() => {
|
|
74419
|
+
hedgeTimer = undefined;
|
|
74420
|
+
if (settled) {
|
|
74421
|
+
return;
|
|
74422
|
+
}
|
|
74423
|
+
const next = peek();
|
|
74424
|
+
if (next !== undefined) {
|
|
74425
|
+
launch(next.leg, next, true);
|
|
74426
|
+
}
|
|
74427
|
+
}, Math.max(0, delay));
|
|
74428
|
+
};
|
|
74429
|
+
/**
|
|
74430
|
+
* Start one attempt.
|
|
74431
|
+
*
|
|
74432
|
+
* @param target The leg to address.
|
|
74433
|
+
* @param candidate The candidate it came from, for a hedge.
|
|
74434
|
+
* @param hedged Whether this is a hedge rather than the leg's first attempt.
|
|
74435
|
+
* @returns Whether it started.
|
|
74436
|
+
*/
|
|
74437
|
+
const launch = (target, candidate, hedged) => {
|
|
74438
|
+
const params = paramsOf(target);
|
|
74439
|
+
const budgetMs = endsAt - now();
|
|
74440
|
+
if (params === undefined || budgetMs <= 0 || isAborted(request.callerSignal)) {
|
|
74441
|
+
return false;
|
|
74442
|
+
}
|
|
74443
|
+
if (candidate !== undefined) {
|
|
74444
|
+
extraLaunched += 1;
|
|
74445
|
+
if (candidate.kind === "equivalent") {
|
|
74446
|
+
nextEquivalent = candidate.index + 1;
|
|
74447
|
+
}
|
|
74448
|
+
}
|
|
74449
|
+
const key = target.route.routeKey;
|
|
74450
|
+
const attempt = {
|
|
74451
|
+
leg: target,
|
|
74452
|
+
handle: startAttempt(target, params, request, budgetMs),
|
|
74453
|
+
startedAt: now(),
|
|
74454
|
+
budgetMs,
|
|
74455
|
+
holdsProbe: breakers.onAttemptStart(key),
|
|
74456
|
+
hedged,
|
|
74457
|
+
attemptIndex: ctx.nextAttemptIndex(),
|
|
74458
|
+
closed: false,
|
|
74459
|
+
};
|
|
74460
|
+
live.add(attempt);
|
|
74461
|
+
if (policy !== undefined) {
|
|
74462
|
+
const tail = quantileOf(target.route, policy.timeoutQuantile);
|
|
74463
|
+
if (tail !== null) {
|
|
74464
|
+
const measuredMs = Math.max(policy.attemptTimeoutFloorMs, policy.kTimeout * tail);
|
|
74465
|
+
if (measuredMs < budgetMs) {
|
|
74466
|
+
attempt.softTimer = setTimeout(() => supersede(attempt, measuredMs), measuredMs);
|
|
74467
|
+
}
|
|
74468
|
+
}
|
|
74469
|
+
}
|
|
74470
|
+
scheduleHedge(attempt);
|
|
74471
|
+
attempt.handle.promise.then((response) => onAnswer(attempt, response), (error) => onFailure(attempt, error));
|
|
74472
|
+
return true;
|
|
74473
|
+
};
|
|
74474
|
+
/**
|
|
74475
|
+
* An attempt ran past its measured timeout: replace it if a same-model
|
|
74476
|
+
* alternative can take over, otherwise leave it running to the leg budget.
|
|
74477
|
+
*
|
|
74478
|
+
* @param attempt The slow attempt.
|
|
74479
|
+
* @param afterMs Its measured timeout.
|
|
74480
|
+
*/
|
|
74481
|
+
const supersede = (attempt, afterMs) => {
|
|
74482
|
+
attempt.softTimer = undefined;
|
|
74483
|
+
if (settled || !live.has(attempt)) {
|
|
74484
|
+
return;
|
|
74485
|
+
}
|
|
74486
|
+
const othersInFlight = live.size > 1;
|
|
74487
|
+
const candidate = othersInFlight ? undefined : peek();
|
|
74488
|
+
if (!othersInFlight && candidate === undefined) {
|
|
74489
|
+
return;
|
|
74490
|
+
}
|
|
74491
|
+
// Recorded now rather than when its rejection arrives, so a replacement
|
|
74492
|
+
// that answers first cannot find it still live and misfile it as a loser.
|
|
74493
|
+
const superseded = new AttemptSupersededError(attempt.leg.route.routeKey, afterMs);
|
|
74494
|
+
close(attempt, superseded);
|
|
74495
|
+
if (candidate !== undefined) {
|
|
74496
|
+
launch(candidate.leg, candidate, true);
|
|
74497
|
+
}
|
|
74498
|
+
finishIfIdle();
|
|
74499
|
+
};
|
|
74500
|
+
/**
|
|
74501
|
+
* Cancel an attempt the chain no longer wants and record it at once, with
|
|
74502
|
+
* no verdict on the provider.
|
|
74503
|
+
*
|
|
74504
|
+
* @param attempt The attempt.
|
|
74505
|
+
* @param reason Why it was cancelled.
|
|
74506
|
+
*/
|
|
74507
|
+
const close = (attempt, reason) => {
|
|
74508
|
+
const fields = baseFields(attempt);
|
|
74509
|
+
attempt.handle.abort(reason);
|
|
74510
|
+
attempt.closed = true;
|
|
74511
|
+
retire(attempt);
|
|
74512
|
+
if (attempt.holdsProbe) {
|
|
74513
|
+
breakers.onAttemptAbandoned(attempt.leg.route.routeKey);
|
|
74514
|
+
}
|
|
74515
|
+
const failure = classify(reason, request.callerSignal);
|
|
74516
|
+
emit(attempt, { ...fields, outcome: failure.outcome, reason: failure.reason });
|
|
74517
|
+
};
|
|
74518
|
+
/**
|
|
74519
|
+
* Stop tracking an attempt and release its breaker bookkeeping.
|
|
74520
|
+
*
|
|
74521
|
+
* @param attempt The attempt.
|
|
74522
|
+
*/
|
|
74523
|
+
const retire = (attempt) => {
|
|
74524
|
+
if (attempt.softTimer !== undefined) {
|
|
74525
|
+
clearTimeout(attempt.softTimer);
|
|
74526
|
+
attempt.softTimer = undefined;
|
|
74527
|
+
}
|
|
74528
|
+
live.delete(attempt);
|
|
74529
|
+
breakers.onAttemptEnd(attempt.leg.route.routeKey);
|
|
74530
|
+
};
|
|
74531
|
+
/**
|
|
74532
|
+
* Record one attempt through the chain.
|
|
74533
|
+
*
|
|
74534
|
+
* @param attempt The attempt.
|
|
74535
|
+
* @param fields What happened.
|
|
74536
|
+
* @param servedProvider The provider's own report of who served, if any.
|
|
74537
|
+
*/
|
|
74538
|
+
const emit = (attempt, fields, servedProvider) => {
|
|
74539
|
+
ctx.record(attempt.leg.route, fields, { hedged: attempt.hedged, attemptIndex: attempt.attemptIndex }, servedProvider);
|
|
74540
|
+
};
|
|
74541
|
+
/**
|
|
74542
|
+
* Base fields shared by every record of an attempt.
|
|
74543
|
+
*
|
|
74544
|
+
* @param attempt The attempt.
|
|
74545
|
+
* @returns The identity and timing fields.
|
|
74546
|
+
*/
|
|
74547
|
+
const baseFields = (attempt) => ({
|
|
74548
|
+
routeKey: attempt.leg.route.routeKey,
|
|
74549
|
+
role: attempt.leg.route.role,
|
|
74550
|
+
provider: attempt.leg.route.providerName,
|
|
74551
|
+
modelId: attempt.leg.route.modelId,
|
|
74552
|
+
durationMs: now() - attempt.startedAt,
|
|
74553
|
+
budgetMs: attempt.budgetMs,
|
|
74554
|
+
});
|
|
74555
|
+
const onAnswer = (attempt, response) => {
|
|
74556
|
+
if (attempt.closed) {
|
|
74557
|
+
return;
|
|
74558
|
+
}
|
|
74559
|
+
const { route } = attempt.leg;
|
|
74560
|
+
const fields = baseFields(attempt);
|
|
74561
|
+
retire(attempt);
|
|
74562
|
+
breakers.onSuccess(route.routeKey, attempt.startedAt);
|
|
74563
|
+
breakers.onLatencySample(route.routeKey, fields.durationMs, route.latencyClass);
|
|
74564
|
+
tracker?.record(route.providerName, route.modelId, promptTokens, fields.durationMs);
|
|
74565
|
+
billed.push(response.usage);
|
|
74566
|
+
if (settled) {
|
|
74567
|
+
// Answered in the same instant as the winner, before its cancellation
|
|
74568
|
+
// arrived. Its spend is real; its answer is not the one returned.
|
|
74569
|
+
emit(attempt, {
|
|
74570
|
+
...fields,
|
|
74571
|
+
outcome: "skipped",
|
|
74572
|
+
reason: "answered after a same-model attempt had already won",
|
|
74573
|
+
servedModel: response.servedModel ?? null,
|
|
74574
|
+
usage: response.usage,
|
|
74575
|
+
}, response.servedProvider);
|
|
74576
|
+
finishIfIdle();
|
|
74577
|
+
return;
|
|
74578
|
+
}
|
|
74579
|
+
settled = true;
|
|
74580
|
+
answer = { response, route, hedged: attempt.hedged };
|
|
74581
|
+
emit(attempt, {
|
|
74582
|
+
...fields,
|
|
74583
|
+
outcome: "ok",
|
|
74584
|
+
servedModel: response.servedModel ?? null,
|
|
74585
|
+
usage: response.usage,
|
|
74586
|
+
}, response.servedProvider);
|
|
74587
|
+
for (const loser of [...live]) {
|
|
74588
|
+
// Cancelled through its own signal and recorded now, so the answer is
|
|
74589
|
+
// returned without waiting for a loser still queued at the provider.
|
|
74590
|
+
close(loser, new HedgeLoserError(loser.leg.route.routeKey));
|
|
74591
|
+
}
|
|
74592
|
+
finishIfIdle();
|
|
74593
|
+
};
|
|
74594
|
+
const onFailure = (attempt, error) => {
|
|
74595
|
+
if (attempt.closed) {
|
|
74596
|
+
return;
|
|
74597
|
+
}
|
|
74598
|
+
const { route } = attempt.leg;
|
|
74599
|
+
const key = route.routeKey;
|
|
74600
|
+
const fields = baseFields(attempt);
|
|
74601
|
+
const hardTimeout = attempt.handle.timedOut();
|
|
74602
|
+
retire(attempt);
|
|
74603
|
+
const failure = classify(attempt.handle.abortReason() ?? error, request.callerSignal);
|
|
74604
|
+
let charged = failure.countsAgainstHealth;
|
|
74605
|
+
if (charged && hardTimeout) {
|
|
74606
|
+
// Every attempt of a leg shares the leg's end, so several can time out
|
|
74607
|
+
// together; the provider is charged once for the leg, as before hedging.
|
|
74608
|
+
charged = !timeoutCharged.has(key);
|
|
74609
|
+
timeoutCharged.add(key);
|
|
74610
|
+
if (budgetIsDeadline) {
|
|
74611
|
+
deadlineBound = true;
|
|
74612
|
+
}
|
|
74613
|
+
}
|
|
74614
|
+
if (charged) {
|
|
74615
|
+
breakers.onFailure(key, failure.failureKind);
|
|
74616
|
+
}
|
|
74617
|
+
else if (attempt.holdsProbe) {
|
|
74618
|
+
// No verdict on the route's health, but the probe slot this attempt
|
|
74619
|
+
// took must come back, or a half-open route admits no probe ever again.
|
|
74620
|
+
breakers.onAttemptAbandoned(key);
|
|
74621
|
+
}
|
|
74622
|
+
if (failure.countsAgainstHealth) {
|
|
74623
|
+
breakers.onLatencySample(key, fields.durationMs, route.latencyClass);
|
|
74624
|
+
}
|
|
74625
|
+
// A provider that answered — with unparseable content, or in prose where a
|
|
74626
|
+
// tool call was mandatory — still billed for the answer.
|
|
74627
|
+
const usage = billedUsageOf(error);
|
|
74628
|
+
if (usage !== undefined) {
|
|
74629
|
+
billed.push(usage);
|
|
74630
|
+
}
|
|
74631
|
+
const answeredBy = error instanceof ToolChoiceIgnoredError ? error.servedModel : undefined;
|
|
74632
|
+
emit(attempt, {
|
|
74633
|
+
...fields,
|
|
74634
|
+
outcome: failure.outcome,
|
|
74635
|
+
reason: failure.reason,
|
|
74636
|
+
...(answeredBy === undefined ? {} : { servedModel: answeredBy }),
|
|
74637
|
+
...(usage === undefined ? {} : { usage }),
|
|
74638
|
+
});
|
|
74639
|
+
if (settled || live.size > 0) {
|
|
74640
|
+
finishIfIdle();
|
|
74641
|
+
return;
|
|
74642
|
+
}
|
|
74643
|
+
if (!hardTimeout && !isAborted(request.callerSignal)) {
|
|
74644
|
+
// A failed attempt with time left moves to the same model at another
|
|
74645
|
+
// provider at once. A duplicate on the provider that just failed is not
|
|
74646
|
+
// started here: it would most likely fail the same way.
|
|
74647
|
+
const candidate = peek();
|
|
74648
|
+
if (candidate !== undefined && candidate.kind === "equivalent") {
|
|
74649
|
+
launch(candidate.leg, candidate, true);
|
|
74650
|
+
}
|
|
74651
|
+
}
|
|
74652
|
+
finishIfIdle();
|
|
74653
|
+
};
|
|
74654
|
+
if (!launch(leg, undefined, false)) {
|
|
74655
|
+
finishIfIdle();
|
|
74656
|
+
}
|
|
74657
|
+
});
|
|
74658
|
+
}
|
|
74659
|
+
|
|
73436
74660
|
/**
|
|
73437
74661
|
* Ordered execution of an alias's fallback chain (PD-3).
|
|
73438
74662
|
*
|
|
@@ -73445,10 +74669,21 @@ function parseStructuredContent(content, responseFormat, usage) {
|
|
|
73445
74669
|
* why the budget is enforced here — at the only place that knows both the
|
|
73446
74670
|
* caller's deadline and how many legs are left to spend it on.
|
|
73447
74671
|
*
|
|
74672
|
+
* Each leg is first run as a group of SAME-MODEL attempts (see `hedge.ts`):
|
|
74673
|
+
* hedged at the model's healthy p90, replaced after a measured timeout, and
|
|
74674
|
+
* reaching the same model at another provider before the leg is given up.
|
|
74675
|
+
* Only then does the walk move to the next leg — and a leg that serves a
|
|
74676
|
+
* different model than the configured one runs only when the caller's
|
|
74677
|
+
* cross-model policy allows it. With no latency evidence and no equivalent
|
|
74678
|
+
* configured, each group is a single attempt with the leg's budget, which is
|
|
74679
|
+
* the serial walk exactly.
|
|
74680
|
+
*
|
|
73448
74681
|
* Nothing here ever substitutes a value for an outcome. When every leg is
|
|
73449
|
-
* exhausted the caller gets a typed error naming each leg and why it failed
|
|
73450
|
-
*
|
|
73451
|
-
*
|
|
74682
|
+
* exhausted the caller gets a typed error naming each leg and why it failed —
|
|
74683
|
+
* `LlmDeadlineExceededError` when the caller's deadline is what ran out,
|
|
74684
|
+
* `ChainExhaustedError` with reason `cross_model_denied` when policy stopped
|
|
74685
|
+
* the walk — because a default returned in place of an answer is a wrong
|
|
74686
|
+
* answer that nobody is told about.
|
|
73452
74687
|
*
|
|
73453
74688
|
* @module llm/fallback-chain
|
|
73454
74689
|
*/
|
|
@@ -73475,21 +74710,64 @@ class ChainExhaustedError extends Error {
|
|
|
73475
74710
|
attempts;
|
|
73476
74711
|
/** Usage spent across the failed attempts, so the spend is still accounted for. */
|
|
73477
74712
|
totalUsage;
|
|
74713
|
+
/**
|
|
74714
|
+
* Why the chain ended. `cross_model_denied`: the configured model's attempts
|
|
74715
|
+
* were spent and policy forbade a different model. Callers map every reason
|
|
74716
|
+
* to no decision; the reason says which remedy applies.
|
|
74717
|
+
*/
|
|
74718
|
+
reason;
|
|
73478
74719
|
/**
|
|
73479
74720
|
* @param alias The alias.
|
|
73480
74721
|
* @param attempts The attempt record.
|
|
73481
74722
|
* @param totalUsage Usage spent across all attempts.
|
|
74723
|
+
* @param reason Why the chain ended; defaults to plain exhaustion.
|
|
73482
74724
|
*/
|
|
73483
|
-
constructor(alias, attempts, totalUsage) {
|
|
74725
|
+
constructor(alias, attempts, totalUsage, reason = "exhausted") {
|
|
73484
74726
|
const detail = attempts
|
|
73485
74727
|
.map((attempt) => `${attempt.role}(${attempt.provider}/${attempt.modelId}): ${attempt.outcome}` +
|
|
73486
74728
|
(attempt.reason === undefined ? "" : ` — ${attempt.reason}`))
|
|
73487
74729
|
.join("; ");
|
|
73488
|
-
|
|
74730
|
+
const why = reason === "cross_model_denied"
|
|
74731
|
+
? " Different-model legs were denied by the caller's cross-model policy."
|
|
74732
|
+
: reason === "deadline_exceeded"
|
|
74733
|
+
? " The caller's deadline ran out."
|
|
74734
|
+
: "";
|
|
74735
|
+
super(`LLM alias "${alias}" exhausted its fallback chain.${why} Attempts: ${detail || "(no leg was servable)"}`);
|
|
73489
74736
|
this.name = "ChainExhaustedError";
|
|
73490
74737
|
this.alias = alias;
|
|
73491
74738
|
this.attempts = attempts;
|
|
73492
74739
|
this.totalUsage = totalUsage;
|
|
74740
|
+
this.reason = reason;
|
|
74741
|
+
}
|
|
74742
|
+
}
|
|
74743
|
+
/**
|
|
74744
|
+
* Thrown when the caller's deadline ran out before any leg answered.
|
|
74745
|
+
*
|
|
74746
|
+
* A subclass of {@link ChainExhaustedError}, so a consumer that already treats
|
|
74747
|
+
* exhaustion as "no answer" keeps doing so, while one that needs to tell "the
|
|
74748
|
+
* models failed" from "we ran out of time" can match this class — the two call
|
|
74749
|
+
* for different remedies (a provider problem versus a budget problem), and
|
|
74750
|
+
* both map to no decision, never to a default.
|
|
74751
|
+
*/
|
|
74752
|
+
class LlmDeadlineExceededError extends ChainExhaustedError {
|
|
74753
|
+
/** Discriminant for consumers that switch on shape rather than class. */
|
|
74754
|
+
kind = "deadline_exceeded";
|
|
74755
|
+
/** The whole-call budget the chain started with, in milliseconds. */
|
|
74756
|
+
deadlineMs;
|
|
74757
|
+
/** The model class of the last attempt dispatched, or null when none was. */
|
|
74758
|
+
lastModelClass;
|
|
74759
|
+
/**
|
|
74760
|
+
* @param alias The alias.
|
|
74761
|
+
* @param attempts The attempt record.
|
|
74762
|
+
* @param totalUsage Usage spent across all attempts.
|
|
74763
|
+
* @param deadlineMs The budget the chain started with.
|
|
74764
|
+
* @param lastModelClass The last dispatched attempt's model class.
|
|
74765
|
+
*/
|
|
74766
|
+
constructor(alias, attempts, totalUsage, deadlineMs, lastModelClass) {
|
|
74767
|
+
super(alias, attempts, totalUsage, "deadline_exceeded");
|
|
74768
|
+
this.name = "LlmDeadlineExceededError";
|
|
74769
|
+
this.deadlineMs = deadlineMs;
|
|
74770
|
+
this.lastModelClass = lastModelClass;
|
|
73493
74771
|
}
|
|
73494
74772
|
}
|
|
73495
74773
|
/**
|
|
@@ -73555,219 +74833,70 @@ function legBudgetMs(routeBudgetMs, deadlineAtMs, nowMs) {
|
|
|
73555
74833
|
}
|
|
73556
74834
|
return Math.min(routeBudgetMs, deadlineAtMs - nowMs);
|
|
73557
74835
|
}
|
|
73558
|
-
/** Raised internally when a leg exceeds its budget. */
|
|
73559
|
-
class LegTimeoutError extends Error {
|
|
73560
|
-
/**
|
|
73561
|
-
* @param routeKey The leg that timed out.
|
|
73562
|
-
* @param budgetMs Its budget in milliseconds.
|
|
73563
|
-
*/
|
|
73564
|
-
constructor(routeKey, budgetMs) {
|
|
73565
|
-
super(`route ${routeKey} exceeded its ${budgetMs} ms budget`);
|
|
73566
|
-
this.name = "LegTimeoutError";
|
|
73567
|
-
}
|
|
73568
|
-
}
|
|
73569
74836
|
/**
|
|
73570
|
-
*
|
|
73571
|
-
*
|
|
73572
|
-
* The timer is always cleared and the abort listener always removed, including
|
|
73573
|
-
* on the success path. A long-lived process that leaked one timer per LLM call
|
|
73574
|
-
* would accumulate them at exactly the rate it does useful work.
|
|
74837
|
+
* The model a leg serves, independent of which provider hosts it.
|
|
73575
74838
|
*
|
|
73576
|
-
* @param
|
|
73577
|
-
* @
|
|
73578
|
-
* @param execution The call context.
|
|
73579
|
-
* @param budgetMs The leg's budget: its route budget cut to the caller's deadline.
|
|
73580
|
-
* @returns The provider's answer.
|
|
74839
|
+
* @param route The leg's route.
|
|
74840
|
+
* @returns Its model class.
|
|
73581
74841
|
*/
|
|
73582
|
-
|
|
73583
|
-
|
|
73584
|
-
const timer = setTimeout(() => {
|
|
73585
|
-
controller.abort(new LegTimeoutError(leg.route.routeKey, budgetMs));
|
|
73586
|
-
}, budgetMs);
|
|
73587
|
-
const forwardAbort = () => {
|
|
73588
|
-
controller.abort(execution.callerSignal?.reason);
|
|
73589
|
-
};
|
|
73590
|
-
if (execution.callerSignal !== undefined) {
|
|
73591
|
-
if (execution.callerSignal.aborted) {
|
|
73592
|
-
forwardAbort();
|
|
73593
|
-
}
|
|
73594
|
-
else {
|
|
73595
|
-
execution.callerSignal.addEventListener("abort", forwardAbort, { once: true });
|
|
73596
|
-
}
|
|
73597
|
-
}
|
|
73598
|
-
try {
|
|
73599
|
-
// The guards wrap the transport rather than the whole leg, so the per-leg
|
|
73600
|
-
// timeout above still bounds the total wait: a caller queued behind the
|
|
73601
|
-
// rate limiter is spending its budget just as surely as one waiting on the
|
|
73602
|
-
// provider, and only one clock should govern both. The leg's own signal is
|
|
73603
|
-
// handed to the guard as well, so a leg whose budget or caller is gone
|
|
73604
|
-
// leaves the queue at once instead of holding its place in it.
|
|
73605
|
-
const response = await withProviderGuards(leg.route.providerName, () => leg.transport.execute({
|
|
73606
|
-
route: leg.route,
|
|
73607
|
-
content: execution.content,
|
|
73608
|
-
responseFormat: execution.responseFormat,
|
|
73609
|
-
params,
|
|
73610
|
-
developerPrompt: execution.developerPrompt,
|
|
73611
|
-
context: execution.context,
|
|
73612
|
-
signal: controller.signal,
|
|
73613
|
-
correlationId: execution.correlationId,
|
|
73614
|
-
}), budgetMs, { modelId: leg.route.modelId, signal: controller.signal });
|
|
73615
|
-
assertToolChoiceHonoured(leg.route, params, response);
|
|
73616
|
-
return response;
|
|
73617
|
-
}
|
|
73618
|
-
finally {
|
|
73619
|
-
clearTimeout(timer);
|
|
73620
|
-
execution.callerSignal?.removeEventListener("abort", forwardAbort);
|
|
73621
|
-
}
|
|
74842
|
+
function modelClassOf(route) {
|
|
74843
|
+
return route.modelClass ?? route.modelId;
|
|
73622
74844
|
}
|
|
73623
74845
|
/**
|
|
73624
|
-
*
|
|
73625
|
-
* rather than that the request or the route is wrong: request timeout, too
|
|
73626
|
-
* early, too many requests, service unavailable, and Anthropic's overloaded.
|
|
73627
|
-
*/
|
|
73628
|
-
const CAPACITY_STATUSES = new Set([408, 425, 429, 503, 529]);
|
|
73629
|
-
/**
|
|
73630
|
-
* Wording providers use for a capacity refusal when the status is lost on the
|
|
73631
|
-
* way (a relayed body, a client library's own error). DeepInfra's is
|
|
73632
|
-
* "Model busy, retry later"; Anthropic's is "Overloaded".
|
|
73633
|
-
*/
|
|
73634
|
-
const CAPACITY_WORDING = /\b(busy|overloaded|capacity|rate[ -]?limit(ed)?|too many requests)\b/i;
|
|
73635
|
-
/**
|
|
73636
|
-
* Whether a failure is the provider saying it is full rather than broken.
|
|
74846
|
+
* Whether a provider-reported model names the model a route addressed.
|
|
73637
74847
|
*
|
|
73638
|
-
*
|
|
73639
|
-
*
|
|
73640
|
-
*
|
|
74848
|
+
* Providers report with or without an organisation prefix and in their own
|
|
74849
|
+
* case, so the comparison is case-insensitive and accepts one side being a
|
|
74850
|
+
* `/`-suffix of the other.
|
|
73641
74851
|
*
|
|
73642
|
-
* @param
|
|
73643
|
-
* @param
|
|
73644
|
-
* @returns Whether
|
|
74852
|
+
* @param reported The provider's report.
|
|
74853
|
+
* @param expected The route's model id.
|
|
74854
|
+
* @returns Whether they name the same model.
|
|
73645
74855
|
*/
|
|
73646
|
-
function
|
|
73647
|
-
|
|
73648
|
-
|
|
73649
|
-
|
|
73650
|
-
return true;
|
|
73651
|
-
}
|
|
73652
|
-
}
|
|
73653
|
-
return CAPACITY_WORDING.test(reason);
|
|
74856
|
+
function isSameReportedModel(reported, expected) {
|
|
74857
|
+
const a = reported.trim().toLowerCase();
|
|
74858
|
+
const b = expected.trim().toLowerCase();
|
|
74859
|
+
return a === b || a.endsWith(`/${b}`) || b.endsWith(`/${a}`);
|
|
73654
74860
|
}
|
|
73655
74861
|
/**
|
|
73656
|
-
*
|
|
74862
|
+
* How an attempt's model relates to the configured one.
|
|
73657
74863
|
*
|
|
73658
|
-
*
|
|
73659
|
-
*
|
|
73660
|
-
*
|
|
73661
|
-
*
|
|
74864
|
+
* A leg addressed to a different model class is `different` whatever it
|
|
74865
|
+
* reported. A leg addressed to the configured class is `same` only when it
|
|
74866
|
+
* answered and the provider named the expected model; `different` when the
|
|
74867
|
+
* provider named another; `unknown` when it answered without saying. An
|
|
74868
|
+
* attempt that never answered carries the relation of the model it was
|
|
74869
|
+
* addressed to.
|
|
73662
74870
|
*
|
|
73663
|
-
*
|
|
73664
|
-
*
|
|
73665
|
-
*
|
|
73666
|
-
*
|
|
73667
|
-
* looks like from outside, and a provider that is actually down still costs no
|
|
73668
|
-
* more than one probe per capacity cooldown.
|
|
73669
|
-
*
|
|
73670
|
-
* @param error The thrown value.
|
|
73671
|
-
* @param callerSignal The caller's cancellation signal, if any.
|
|
73672
|
-
* @returns The outcome and whether it counts against route health.
|
|
74871
|
+
* @param route The attempt's route.
|
|
74872
|
+
* @param configuredClass The configured model class.
|
|
74873
|
+
* @param fields What the attempt recorded.
|
|
74874
|
+
* @returns The relation.
|
|
73673
74875
|
*/
|
|
73674
|
-
function
|
|
73675
|
-
if (
|
|
73676
|
-
return
|
|
73677
|
-
outcome: "skipped",
|
|
73678
|
-
reason: "caller cancelled",
|
|
73679
|
-
countsAgainstHealth: false,
|
|
73680
|
-
failureKind: "hard",
|
|
73681
|
-
};
|
|
73682
|
-
}
|
|
73683
|
-
if (error instanceof LegTimeoutError) {
|
|
73684
|
-
return {
|
|
73685
|
-
outcome: "timeout",
|
|
73686
|
-
reason: error.message,
|
|
73687
|
-
countsAgainstHealth: true,
|
|
73688
|
-
failureKind: "capacity",
|
|
73689
|
-
};
|
|
73690
|
-
}
|
|
73691
|
-
if (error instanceof UnsupportedCapabilityError) {
|
|
73692
|
-
return {
|
|
73693
|
-
outcome: "skipped",
|
|
73694
|
-
reason: error.message,
|
|
73695
|
-
countsAgainstHealth: false,
|
|
73696
|
-
failureKind: "hard",
|
|
73697
|
-
};
|
|
73698
|
-
}
|
|
73699
|
-
if (error instanceof ToolChoiceIgnoredError) {
|
|
73700
|
-
// The route answered; it broke a declared guarantee rather than failing to
|
|
73701
|
-
// be available, so its breaker is not charged for it.
|
|
73702
|
-
return {
|
|
73703
|
-
outcome: "error",
|
|
73704
|
-
reason: error.message,
|
|
73705
|
-
countsAgainstHealth: false,
|
|
73706
|
-
failureKind: "hard",
|
|
73707
|
-
};
|
|
74876
|
+
function modelClassRelationOf(route, configuredClass, fields) {
|
|
74877
|
+
if (modelClassOf(route) !== configuredClass) {
|
|
74878
|
+
return "different";
|
|
73708
74879
|
}
|
|
73709
|
-
|
|
73710
|
-
|
|
73711
|
-
// permanently reroute traffic nobody chose to reroute.
|
|
73712
|
-
if (error instanceof RateGuardTimeoutError) {
|
|
73713
|
-
return {
|
|
73714
|
-
outcome: "skipped",
|
|
73715
|
-
reason: error.message,
|
|
73716
|
-
countsAgainstHealth: false,
|
|
73717
|
-
failureKind: "hard",
|
|
73718
|
-
};
|
|
74880
|
+
if (fields.outcome !== "ok") {
|
|
74881
|
+
return "same";
|
|
73719
74882
|
}
|
|
73720
|
-
|
|
73721
|
-
|
|
73722
|
-
return {
|
|
73723
|
-
outcome: "timeout",
|
|
73724
|
-
reason: `aborted: ${reason}`,
|
|
73725
|
-
countsAgainstHealth: true,
|
|
73726
|
-
failureKind: "capacity",
|
|
73727
|
-
};
|
|
74883
|
+
if (fields.servedModel === undefined || fields.servedModel === null) {
|
|
74884
|
+
return "unknown";
|
|
73728
74885
|
}
|
|
73729
|
-
|
|
73730
|
-
// The provider answered, badly. That is a route defect, not a full queue,
|
|
73731
|
-
// whatever words the unparseable content happens to contain.
|
|
73732
|
-
return { outcome: "error", reason, countsAgainstHealth: true, failureKind: "hard" };
|
|
73733
|
-
}
|
|
73734
|
-
return {
|
|
73735
|
-
outcome: "error",
|
|
73736
|
-
reason,
|
|
73737
|
-
countsAgainstHealth: true,
|
|
73738
|
-
failureKind: isCapacitySignal(error, reason) ? "capacity" : "hard",
|
|
73739
|
-
};
|
|
73740
|
-
}
|
|
73741
|
-
/**
|
|
73742
|
-
* Whether the caller has stopped waiting.
|
|
73743
|
-
*
|
|
73744
|
-
* Read through a function rather than inline, because `AbortSignal.aborted` is
|
|
73745
|
-
* a live getter: it can flip to true while a leg is in flight, but a compiler
|
|
73746
|
-
* that narrowed it at the top of the loop would prove the later check
|
|
73747
|
-
* unreachable and invite its removal. The check is not redundant — it is the
|
|
73748
|
-
* only thing that stops the chain spending money on an answer nobody will read.
|
|
73749
|
-
*
|
|
73750
|
-
* @param signal The caller's signal, if any.
|
|
73751
|
-
* @returns Whether the call has been cancelled.
|
|
73752
|
-
*/
|
|
73753
|
-
function isAborted(signal) {
|
|
73754
|
-
return signal !== undefined && signal.aborted;
|
|
74886
|
+
return isSameReportedModel(fields.servedModel, route.modelId) ? "same" : "different";
|
|
73755
74887
|
}
|
|
73756
74888
|
/**
|
|
73757
|
-
* The
|
|
74889
|
+
* The prompt size the latency evidence is bucketed by.
|
|
73758
74890
|
*
|
|
73759
|
-
*
|
|
73760
|
-
*
|
|
73761
|
-
* never answered (timeout, outage, skip) carries no usage, and none is invented.
|
|
73762
|
-
*
|
|
73763
|
-
* @param error The thrown value.
|
|
73764
|
-
* @returns The billed usage, or undefined when the leg never produced an answer.
|
|
74891
|
+
* @param execution The call.
|
|
74892
|
+
* @returns Estimated prompt tokens, or null.
|
|
73765
74893
|
*/
|
|
73766
|
-
function
|
|
73767
|
-
|
|
73768
|
-
|
|
73769
|
-
|
|
73770
|
-
|
|
74894
|
+
function promptTokensOf(execution) {
|
|
74895
|
+
return estimatePromptTokens([
|
|
74896
|
+
execution.content,
|
|
74897
|
+
execution.developerPrompt,
|
|
74898
|
+
execution.context,
|
|
74899
|
+
]);
|
|
73771
74900
|
}
|
|
73772
74901
|
/**
|
|
73773
74902
|
* Walk a chain until a leg answers.
|
|
@@ -73775,12 +74904,58 @@ function billedUsageOf(error) {
|
|
|
73775
74904
|
* @param alias The alias being served, for error attribution.
|
|
73776
74905
|
* @param execution The call context.
|
|
73777
74906
|
* @returns The first successful leg's answer, with the full attempt record.
|
|
74907
|
+
* @throws {LlmDeadlineExceededError} When the caller's deadline ran out first.
|
|
73778
74908
|
* @throws {ChainExhaustedError} When no leg produced an answer.
|
|
73779
74909
|
*/
|
|
73780
74910
|
async function executeChain(alias, execution) {
|
|
73781
74911
|
const now = execution.now ?? Date.now;
|
|
74912
|
+
const startedAt = now();
|
|
73782
74913
|
const attempts = [];
|
|
74914
|
+
const firstRoute = execution.legs[0]?.route;
|
|
74915
|
+
const configuredClass = execution.configuredModelClass ?? (firstRoute === undefined ? "" : modelClassOf(firstRoute));
|
|
74916
|
+
const policy = execution.crossModelPolicy ?? "allow_record";
|
|
74917
|
+
const promptTokens = execution.latency === undefined ? null : promptTokensOf(execution);
|
|
73783
74918
|
let totalUsage = EMPTY_USAGE;
|
|
74919
|
+
let dispatchIndex = 0;
|
|
74920
|
+
let deadlineHit = false;
|
|
74921
|
+
let crossModelDenied = false;
|
|
74922
|
+
let lastModelClass = null;
|
|
74923
|
+
/**
|
|
74924
|
+
* Record one attempt with its provenance.
|
|
74925
|
+
*
|
|
74926
|
+
* @param route The attempt's route.
|
|
74927
|
+
* @param fields What happened.
|
|
74928
|
+
* @param dispatch Whether it was a hedge, and its dispatch index.
|
|
74929
|
+
* @param servedProvider The provider's own report of who served, if any.
|
|
74930
|
+
* @returns The record.
|
|
74931
|
+
*/
|
|
74932
|
+
const record = (route, fields, dispatch, servedProvider) => {
|
|
74933
|
+
const full = {
|
|
74934
|
+
...fields,
|
|
74935
|
+
servedProvider: servedProvider ?? route.providerName,
|
|
74936
|
+
modelClass: modelClassOf(route),
|
|
74937
|
+
modelClassRelation: modelClassRelationOf(route, configuredClass, fields),
|
|
74938
|
+
...(dispatch === undefined
|
|
74939
|
+
? {}
|
|
74940
|
+
: { hedged: dispatch.hedged, attemptIndex: dispatch.attemptIndex }),
|
|
74941
|
+
};
|
|
74942
|
+
attempts.push(full);
|
|
74943
|
+
execution.onAttempt?.(full);
|
|
74944
|
+
return full;
|
|
74945
|
+
};
|
|
74946
|
+
/**
|
|
74947
|
+
* The identity fields of a leg that was never dispatched.
|
|
74948
|
+
*
|
|
74949
|
+
* @param route The leg's route.
|
|
74950
|
+
* @returns The fields.
|
|
74951
|
+
*/
|
|
74952
|
+
const undispatched = (route) => ({
|
|
74953
|
+
routeKey: route.routeKey,
|
|
74954
|
+
role: route.role,
|
|
74955
|
+
provider: route.providerName,
|
|
74956
|
+
modelId: route.modelId,
|
|
74957
|
+
durationMs: 0,
|
|
74958
|
+
});
|
|
73784
74959
|
for (const leg of execution.legs) {
|
|
73785
74960
|
const { route } = leg;
|
|
73786
74961
|
if (isAborted(execution.callerSignal)) {
|
|
@@ -73788,108 +74963,80 @@ async function executeChain(alias, execution) {
|
|
|
73788
74963
|
// spend money on an answer nobody will read.
|
|
73789
74964
|
break;
|
|
73790
74965
|
}
|
|
73791
|
-
if (
|
|
73792
|
-
|
|
73793
|
-
|
|
73794
|
-
|
|
73795
|
-
provider: route.providerName,
|
|
73796
|
-
modelId: route.modelId,
|
|
74966
|
+
if (policy === "deny" && modelClassOf(route) !== configuredClass) {
|
|
74967
|
+
crossModelDenied = true;
|
|
74968
|
+
record(route, {
|
|
74969
|
+
...undispatched(route),
|
|
73797
74970
|
outcome: "skipped",
|
|
73798
|
-
|
|
73799
|
-
|
|
73800
|
-
|
|
73801
|
-
|
|
73802
|
-
|
|
74971
|
+
reason: `cross-model leg denied by policy: configured model is ${configuredClass}, this leg serves ${modelClassOf(route)}`,
|
|
74972
|
+
});
|
|
74973
|
+
continue;
|
|
74974
|
+
}
|
|
74975
|
+
if (leg.params instanceof UnsupportedCapabilityError) {
|
|
74976
|
+
record(route, { ...undispatched(route), outcome: "skipped", reason: leg.params.message });
|
|
73803
74977
|
continue;
|
|
73804
74978
|
}
|
|
73805
74979
|
if (!execution.breakers.allows(route.routeKey)) {
|
|
73806
|
-
|
|
73807
|
-
|
|
73808
|
-
role: route.role,
|
|
73809
|
-
provider: route.providerName,
|
|
73810
|
-
modelId: route.modelId,
|
|
74980
|
+
record(route, {
|
|
74981
|
+
...undispatched(route),
|
|
73811
74982
|
outcome: "breaker-open",
|
|
73812
|
-
durationMs: 0,
|
|
73813
74983
|
reason: `circuit breaker is ${execution.breakers.stateOf(route.routeKey)}`,
|
|
73814
|
-
};
|
|
73815
|
-
attempts.push(record);
|
|
73816
|
-
execution.onAttempt?.(record);
|
|
74984
|
+
});
|
|
73817
74985
|
continue;
|
|
73818
74986
|
}
|
|
73819
74987
|
const budgetMs = legBudgetMs(route.timeoutMs, execution.deadlineAtMs, now());
|
|
73820
74988
|
if (budgetMs <= 0) {
|
|
73821
74989
|
// The caller's deadline is spent. Dispatching now would start a call
|
|
73822
74990
|
// that is cancelled the moment it begins, and charge nothing but noise.
|
|
73823
|
-
|
|
73824
|
-
|
|
73825
|
-
|
|
73826
|
-
provider: route.providerName,
|
|
73827
|
-
modelId: route.modelId,
|
|
74991
|
+
deadlineHit = true;
|
|
74992
|
+
record(route, {
|
|
74993
|
+
...undispatched(route),
|
|
73828
74994
|
outcome: "skipped",
|
|
73829
|
-
durationMs: 0,
|
|
73830
74995
|
reason: "caller deadline exhausted before this leg",
|
|
73831
|
-
};
|
|
73832
|
-
attempts.push(record);
|
|
73833
|
-
execution.onAttempt?.(record);
|
|
74996
|
+
});
|
|
73834
74997
|
continue;
|
|
73835
74998
|
}
|
|
73836
|
-
|
|
73837
|
-
const
|
|
73838
|
-
|
|
73839
|
-
|
|
73840
|
-
|
|
73841
|
-
|
|
73842
|
-
|
|
73843
|
-
|
|
73844
|
-
|
|
73845
|
-
|
|
73846
|
-
|
|
73847
|
-
|
|
73848
|
-
|
|
73849
|
-
|
|
73850
|
-
|
|
73851
|
-
|
|
73852
|
-
}
|
|
73853
|
-
|
|
73854
|
-
|
|
73855
|
-
|
|
74999
|
+
lastModelClass = modelClassOf(route);
|
|
75000
|
+
const group = await runSameModelGroup(leg, budgetMs, budgetMs < route.timeoutMs, {
|
|
75001
|
+
request: execution,
|
|
75002
|
+
breakers: execution.breakers,
|
|
75003
|
+
now,
|
|
75004
|
+
policy: execution.hedging,
|
|
75005
|
+
tracker: execution.latency,
|
|
75006
|
+
promptTokens,
|
|
75007
|
+
admitDuplicate: execution.admitDuplicate ?? (() => false),
|
|
75008
|
+
record: (attemptRoute, fields, dispatch, servedProvider) => {
|
|
75009
|
+
record(attemptRoute, fields, dispatch, servedProvider);
|
|
75010
|
+
},
|
|
75011
|
+
nextAttemptIndex: () => {
|
|
75012
|
+
const index = dispatchIndex;
|
|
75013
|
+
dispatchIndex += 1;
|
|
75014
|
+
return index;
|
|
75015
|
+
},
|
|
75016
|
+
});
|
|
75017
|
+
for (const usage of group.billed) {
|
|
75018
|
+
totalUsage = sumUsage(totalUsage, usage);
|
|
73856
75019
|
}
|
|
73857
|
-
|
|
73858
|
-
|
|
73859
|
-
|
|
73860
|
-
|
|
73861
|
-
|
|
73862
|
-
|
|
73863
|
-
|
|
73864
|
-
|
|
73865
|
-
|
|
73866
|
-
|
|
73867
|
-
// A provider that answered — with unparseable content, or in prose where a
|
|
73868
|
-
// tool call was mandatory — still billed for the answer; the spend belongs
|
|
73869
|
-
// in the total whether or not a later leg serves.
|
|
73870
|
-
const billed = billedUsageOf(error);
|
|
73871
|
-
const answeredBy = error instanceof ToolChoiceIgnoredError ? error.servedModel : undefined;
|
|
73872
|
-
totalUsage = sumUsage(totalUsage, billed);
|
|
73873
|
-
const record = {
|
|
73874
|
-
routeKey: route.routeKey,
|
|
73875
|
-
role: route.role,
|
|
73876
|
-
provider: route.providerName,
|
|
73877
|
-
modelId: route.modelId,
|
|
73878
|
-
outcome,
|
|
73879
|
-
durationMs: now() - startedAt,
|
|
73880
|
-
budgetMs,
|
|
73881
|
-
reason,
|
|
73882
|
-
...(answeredBy === undefined ? {} : { servedModel: answeredBy }),
|
|
73883
|
-
...(billed === undefined ? {} : { usage: billed }),
|
|
75020
|
+
deadlineHit = deadlineHit || group.deadlineBound;
|
|
75021
|
+
if (group.answer !== undefined) {
|
|
75022
|
+
const winner = attempts.find((attempt) => attempt.outcome === "ok" && attempt.routeKey === group.answer?.route.routeKey);
|
|
75023
|
+
return {
|
|
75024
|
+
response: group.answer.response,
|
|
75025
|
+
servedBy: group.answer.route,
|
|
75026
|
+
attempts,
|
|
75027
|
+
totalUsage,
|
|
75028
|
+
modelClassRelation: winner?.modelClassRelation ?? "unknown",
|
|
75029
|
+
hedged: group.answer.hedged,
|
|
73884
75030
|
};
|
|
73885
|
-
attempts.push(record);
|
|
73886
|
-
execution.onAttempt?.(record);
|
|
73887
|
-
if (outcome === "skipped" && isAborted(execution.callerSignal)) {
|
|
73888
|
-
break;
|
|
73889
|
-
}
|
|
73890
75031
|
}
|
|
75032
|
+
if (isAborted(execution.callerSignal)) {
|
|
75033
|
+
break;
|
|
75034
|
+
}
|
|
75035
|
+
}
|
|
75036
|
+
if (deadlineHit && !isAborted(execution.callerSignal) && execution.deadlineAtMs !== undefined) {
|
|
75037
|
+
throw new LlmDeadlineExceededError(alias, attempts, totalUsage, execution.deadlineAtMs - startedAt, lastModelClass);
|
|
73891
75038
|
}
|
|
73892
|
-
throw new ChainExhaustedError(alias, attempts, totalUsage);
|
|
75039
|
+
throw new ChainExhaustedError(alias, attempts, totalUsage, crossModelDenied ? "cross_model_denied" : "exhausted");
|
|
73893
75040
|
}
|
|
73894
75041
|
|
|
73895
75042
|
var schema_version = 1;
|
|
@@ -73906,7 +75053,37 @@ var defaults = {
|
|
|
73906
75053
|
failure_threshold: 5,
|
|
73907
75054
|
cooldown_ms: 60000,
|
|
73908
75055
|
capacity_cooldown_ms: 15000,
|
|
73909
|
-
half_open_probes: 1
|
|
75056
|
+
half_open_probes: 1,
|
|
75057
|
+
probe_fraction: 0.1,
|
|
75058
|
+
latency_trip: {
|
|
75059
|
+
enabled: false,
|
|
75060
|
+
slo_ms: {
|
|
75061
|
+
"hot-path": 20000,
|
|
75062
|
+
background: 60000,
|
|
75063
|
+
batch: 240000
|
|
75064
|
+
},
|
|
75065
|
+
quantile: 0.9,
|
|
75066
|
+
window_size: 20,
|
|
75067
|
+
trip_windows: 3,
|
|
75068
|
+
of_windows: 5
|
|
75069
|
+
}
|
|
75070
|
+
},
|
|
75071
|
+
hedging: {
|
|
75072
|
+
max_same_model_hedges: 1,
|
|
75073
|
+
hedge_quantile: 0.9,
|
|
75074
|
+
timeout_quantile: 0.99,
|
|
75075
|
+
k_timeout: 3,
|
|
75076
|
+
attempt_timeout_floor_ms: 5000,
|
|
75077
|
+
max_attempt_share: 0.5,
|
|
75078
|
+
duplicate_headroom_reserve: 0.25,
|
|
75079
|
+
min_samples: 20,
|
|
75080
|
+
window_size: 200,
|
|
75081
|
+
sample_max_age_ms: 900000,
|
|
75082
|
+
prompt_token_buckets: [
|
|
75083
|
+
2000,
|
|
75084
|
+
8000,
|
|
75085
|
+
32000
|
|
75086
|
+
]
|
|
73910
75087
|
}
|
|
73911
75088
|
};
|
|
73912
75089
|
var providers = {
|
|
@@ -74229,10 +75406,10 @@ var aliases = {
|
|
|
74229
75406
|
context_window: 1000000
|
|
74230
75407
|
},
|
|
74231
75408
|
price_per_mtok: {
|
|
74232
|
-
input:
|
|
74233
|
-
output:
|
|
74234
|
-
as_of: "2026-09-
|
|
74235
|
-
source: "
|
|
75409
|
+
input: 5,
|
|
75410
|
+
output: 25,
|
|
75411
|
+
as_of: "2026-09-30",
|
|
75412
|
+
source: "lumic-utils/src/functions/llm-config.ts anthropicModelCosts['claude-opus-4-7'] = 5/25, the Opus-tier rate. CORRECTED from 10/50, which is the claude-fable-5 rate: this leg carried a Fable-class anchor for an Opus-class model, so the same model id was priced 2x apart on llm.reason and llm.agentic and the declared cost of every alias's revert target depended on which alias a reader happened to look at."
|
|
74236
75413
|
}
|
|
74237
75414
|
}
|
|
74238
75415
|
]
|
|
@@ -74642,6 +75819,86 @@ const ISOLATED_SUFFIX = ".isolated";
|
|
|
74642
75819
|
* caller change every other caller's routing.
|
|
74643
75820
|
*/
|
|
74644
75821
|
const routeTable = rawTable;
|
|
75822
|
+
/** Separates a leg's route key from the provider of one of its equivalents. */
|
|
75823
|
+
const EQUIVALENT_SEPARATOR = "~";
|
|
75824
|
+
/** Gateway name segment for an equivalent leg. */
|
|
75825
|
+
const EQUIVALENT_NAMESPACE = "equivalent";
|
|
75826
|
+
/**
|
|
75827
|
+
* Whether a number lies in a closed or open range.
|
|
75828
|
+
*
|
|
75829
|
+
* @param value The value.
|
|
75830
|
+
* @param min Lower bound.
|
|
75831
|
+
* @param max Upper bound.
|
|
75832
|
+
* @param open Whether the bounds are excluded.
|
|
75833
|
+
* @returns Whether it is in range.
|
|
75834
|
+
*/
|
|
75835
|
+
function inRange(value, min, max, open) {
|
|
75836
|
+
if (typeof value !== "number" || !Number.isFinite(value)) {
|
|
75837
|
+
return false;
|
|
75838
|
+
}
|
|
75839
|
+
return open ? value > min && value < max : value >= min && value <= max;
|
|
75840
|
+
}
|
|
75841
|
+
/**
|
|
75842
|
+
* Bounds violations in the tail-latency defaults (hedging, probe scaling and
|
|
75843
|
+
* the latency trip).
|
|
75844
|
+
*
|
|
75845
|
+
* These values are mechanics rather than routing choices, but a value outside
|
|
75846
|
+
* its bounds turns a mechanic into a routing change — a quantile of 1.0 makes
|
|
75847
|
+
* every hedge wait for the slowest answer ever seen, a zero floor lets a
|
|
75848
|
+
* measured timeout cut an attempt the instant it starts — so they are checked
|
|
75849
|
+
* as strictly as the routing itself.
|
|
75850
|
+
*
|
|
75851
|
+
* @param defaults The table's defaults.
|
|
75852
|
+
* @returns One message per violation; empty when every value is in bounds.
|
|
75853
|
+
*/
|
|
75854
|
+
function tailLatencyViolations(defaults) {
|
|
75855
|
+
const violations = [];
|
|
75856
|
+
const check = (ok, message) => {
|
|
75857
|
+
if (!ok) {
|
|
75858
|
+
violations.push(message);
|
|
75859
|
+
}
|
|
75860
|
+
};
|
|
75861
|
+
const hedging = defaults.hedging;
|
|
75862
|
+
if (hedging !== undefined) {
|
|
75863
|
+
check(Number.isInteger(hedging.max_same_model_hedges) && inRange(hedging.max_same_model_hedges, 0, 3, false), "hedging.max_same_model_hedges must be an integer in [0, 3]");
|
|
75864
|
+
check(inRange(hedging.hedge_quantile, 0.5, 1, true), "hedging.hedge_quantile must be in (0.5, 1)");
|
|
75865
|
+
check(inRange(hedging.timeout_quantile, 0.5, 1, true), "hedging.timeout_quantile must be in (0.5, 1)");
|
|
75866
|
+
check(hedging.timeout_quantile >= hedging.hedge_quantile, "hedging.timeout_quantile must not be below hedging.hedge_quantile");
|
|
75867
|
+
check(inRange(hedging.k_timeout, 1, 10, false), "hedging.k_timeout must be in [1, 10]");
|
|
75868
|
+
check(inRange(hedging.attempt_timeout_floor_ms, 1000, Number.MAX_SAFE_INTEGER, false), "hedging.attempt_timeout_floor_ms must be at least 1000");
|
|
75869
|
+
check(inRange(hedging.max_attempt_share, 0.25, 1, false), "hedging.max_attempt_share must be in [0.25, 1]");
|
|
75870
|
+
check(inRange(hedging.duplicate_headroom_reserve, 0, 0.9, false), "hedging.duplicate_headroom_reserve must be in [0, 0.9]");
|
|
75871
|
+
check(Number.isInteger(hedging.min_samples) && inRange(hedging.min_samples, 5, 10_000, false), "hedging.min_samples must be an integer in [5, 10000]");
|
|
75872
|
+
check(Number.isInteger(hedging.window_size) && hedging.window_size >= hedging.min_samples, "hedging.window_size must be an integer no smaller than min_samples");
|
|
75873
|
+
check(inRange(hedging.sample_max_age_ms, 60_000, Number.MAX_SAFE_INTEGER, false), "hedging.sample_max_age_ms must be at least 60000");
|
|
75874
|
+
check(hedging.prompt_token_buckets.every((edge, index, edges) => Number.isFinite(edge) && edge > 0 && (index === 0 || edge > edges[index - 1])), "hedging.prompt_token_buckets must be positive and strictly ascending");
|
|
75875
|
+
}
|
|
75876
|
+
const breaker = defaults.circuit_breaker;
|
|
75877
|
+
if (breaker.probe_fraction !== undefined) {
|
|
75878
|
+
check(inRange(breaker.probe_fraction, 0, 0.5, false), "circuit_breaker.probe_fraction must be in [0, 0.5]");
|
|
75879
|
+
}
|
|
75880
|
+
const trip = breaker.latency_trip;
|
|
75881
|
+
if (trip !== undefined) {
|
|
75882
|
+
check(inRange(trip.quantile, 0.5, 1, true), "circuit_breaker.latency_trip.quantile must be in (0.5, 1)");
|
|
75883
|
+
check(Number.isInteger(trip.window_size) && trip.window_size >= 5, "circuit_breaker.latency_trip.window_size must be an integer of at least 5");
|
|
75884
|
+
check(Number.isInteger(trip.of_windows) &&
|
|
75885
|
+
Number.isInteger(trip.trip_windows) &&
|
|
75886
|
+
trip.trip_windows >= 1 &&
|
|
75887
|
+
trip.trip_windows <= trip.of_windows, "circuit_breaker.latency_trip needs integer 1 <= trip_windows <= of_windows");
|
|
75888
|
+
for (const latencyClass of Object.keys(defaults.request_timeout_ms)) {
|
|
75889
|
+
check(inRange(trip.slo_ms[latencyClass], 1000, defaults.request_timeout_ms[latencyClass], false), `circuit_breaker.latency_trip.slo_ms.${latencyClass} must be in [1000, its request timeout]`);
|
|
75890
|
+
}
|
|
75891
|
+
}
|
|
75892
|
+
return violations;
|
|
75893
|
+
}
|
|
75894
|
+
{
|
|
75895
|
+
// Checked when the table loads: it is bundled, so a violation can only
|
|
75896
|
+
// arrive in a release, and it must stop that release rather than run it.
|
|
75897
|
+
const violations = tailLatencyViolations(routeTable.defaults);
|
|
75898
|
+
if (violations.length > 0) {
|
|
75899
|
+
throw new Error(`alias route table has out-of-bounds tail-latency defaults: ${violations.join("; ")}`);
|
|
75900
|
+
}
|
|
75901
|
+
}
|
|
74645
75902
|
/**
|
|
74646
75903
|
* Every alias the table defines.
|
|
74647
75904
|
*
|
|
@@ -74821,6 +76078,8 @@ function resolveChain(alias, options = {}) {
|
|
|
74821
76078
|
});
|
|
74822
76079
|
continue;
|
|
74823
76080
|
}
|
|
76081
|
+
const routeKey = routeKeyFor(alias, isolated, route.role);
|
|
76082
|
+
const modelClass = route.model_class ?? admission.modelId;
|
|
74824
76083
|
resolved.push({
|
|
74825
76084
|
alias,
|
|
74826
76085
|
isolated,
|
|
@@ -74830,13 +76089,67 @@ function resolveChain(alias, options = {}) {
|
|
|
74830
76089
|
modelId: admission.modelId,
|
|
74831
76090
|
lumicModel: route.lumic_model ?? null,
|
|
74832
76091
|
params: route.params ?? {},
|
|
74833
|
-
routeKey
|
|
76092
|
+
routeKey,
|
|
74834
76093
|
timeoutMs,
|
|
74835
|
-
|
|
76094
|
+
modelClass,
|
|
76095
|
+
latencyClass: definition.latency_class,
|
|
76096
|
+
equivalents: resolveEquivalents(route, {
|
|
76097
|
+
alias,
|
|
76098
|
+
isolated,
|
|
76099
|
+
routeKey,
|
|
76100
|
+
timeoutMs,
|
|
76101
|
+
modelClass,
|
|
76102
|
+
latencyClass: definition.latency_class,
|
|
76103
|
+
}),
|
|
74836
76104
|
});
|
|
74837
76105
|
}
|
|
74838
76106
|
return { alias, isolated, routes: resolved, exclusions };
|
|
74839
76107
|
}
|
|
76108
|
+
/**
|
|
76109
|
+
* Resolve a leg's same-model equivalents that can serve today.
|
|
76110
|
+
*
|
|
76111
|
+
* An equivalent is admitted by the leg's own rules — a confirmed model id and a
|
|
76112
|
+
* live provider account — and inherits the leg's model class, because being the
|
|
76113
|
+
* same model is what makes it an equivalent. One that cannot serve is simply
|
|
76114
|
+
* absent: it is never a reason to skip the leg.
|
|
76115
|
+
*
|
|
76116
|
+
* @param route The authored leg.
|
|
76117
|
+
* @param leg The resolved leg's identity.
|
|
76118
|
+
* @param leg.alias The alias.
|
|
76119
|
+
* @param leg.isolated Whether this is the isolated variant.
|
|
76120
|
+
* @param leg.routeKey The leg's route key.
|
|
76121
|
+
* @param leg.timeoutMs The leg's budget.
|
|
76122
|
+
* @param leg.modelClass The leg's model class.
|
|
76123
|
+
* @param leg.latencyClass The alias's latency class.
|
|
76124
|
+
* @returns The servable equivalents, in table order.
|
|
76125
|
+
*/
|
|
76126
|
+
function resolveEquivalents(route, leg) {
|
|
76127
|
+
const resolved = [];
|
|
76128
|
+
for (const equivalent of route.equivalents ?? []) {
|
|
76129
|
+
const provider = routeTable.providers[equivalent.provider];
|
|
76130
|
+
if (provider === undefined ||
|
|
76131
|
+
provider.account_status !== "live" ||
|
|
76132
|
+
equivalent.model_id_status !== "confirmed" ||
|
|
76133
|
+
equivalent.model_id === null) {
|
|
76134
|
+
continue;
|
|
76135
|
+
}
|
|
76136
|
+
resolved.push({
|
|
76137
|
+
alias: leg.alias,
|
|
76138
|
+
isolated: leg.isolated,
|
|
76139
|
+
role: route.role,
|
|
76140
|
+
providerName: equivalent.provider,
|
|
76141
|
+
provider,
|
|
76142
|
+
modelId: equivalent.model_id,
|
|
76143
|
+
lumicModel: equivalent.lumic_model ?? null,
|
|
76144
|
+
params: equivalent.params ?? route.params ?? {},
|
|
76145
|
+
routeKey: `${leg.routeKey}${EQUIVALENT_SEPARATOR}${equivalent.provider}`,
|
|
76146
|
+
timeoutMs: leg.timeoutMs,
|
|
76147
|
+
modelClass: leg.modelClass,
|
|
76148
|
+
latencyClass: leg.latencyClass,
|
|
76149
|
+
});
|
|
76150
|
+
}
|
|
76151
|
+
return resolved;
|
|
76152
|
+
}
|
|
74840
76153
|
/**
|
|
74841
76154
|
* The permanent closed-incumbent leg of an alias (PD-11).
|
|
74842
76155
|
*
|
|
@@ -74865,6 +76178,12 @@ function closedIncumbentLeg(chain) {
|
|
|
74865
76178
|
*/
|
|
74866
76179
|
function gatewayModelNameFor(route, chain) {
|
|
74867
76180
|
const base = `${route.alias}${route.isolated ? ISOLATED_SUFFIX : ""}`;
|
|
76181
|
+
if (route.routeKey.includes(EQUIVALENT_SEPARATOR)) {
|
|
76182
|
+
// The same model at another provider is its own gateway deployment, named
|
|
76183
|
+
// so the gateway serves exactly that deployment and nothing it might fall
|
|
76184
|
+
// back to on its own.
|
|
76185
|
+
return `${base}.${EQUIVALENT_NAMESPACE}.${route.role}.${route.providerName}`;
|
|
76186
|
+
}
|
|
74868
76187
|
return chain.routes[0]?.routeKey === route.routeKey
|
|
74869
76188
|
? base
|
|
74870
76189
|
: `${base}.fallback.${route.role}`;
|
|
@@ -75152,8 +76471,11 @@ async function resolveDefaultDirectCaller() {
|
|
|
75152
76471
|
* gateway by its own model name, so the gateway serves one deployment per leg
|
|
75153
76472
|
* and needs no fallback of its own. A proxy-side fallback inside a leg would
|
|
75154
76473
|
* spend the leg's budget on a model the chain did not choose and report the
|
|
75155
|
-
* answer as the leg's; the served model is read from the
|
|
75156
|
-
*
|
|
76474
|
+
* answer as the leg's; the served model is read from the gateway's own
|
|
76475
|
+
* served-model header, else from the upstream body's `model`, so such a
|
|
76476
|
+
* substitution stays visible while any remains configured. A body `model` that
|
|
76477
|
+
* merely echoes the name the leg was addressed by is the gateway naming its
|
|
76478
|
+
* model GROUP, not the model that answered, and is reported as unknown.
|
|
75157
76479
|
*
|
|
75158
76480
|
* The gateway key is read from the environment by NAME at call time and never
|
|
75159
76481
|
* stored, logged, or included in an error (PD-2). Reading it per call rather
|
|
@@ -75169,6 +76491,14 @@ const ERROR_BODY_EXCERPT = 400;
|
|
|
75169
76491
|
const RESPONSE_COST_HEADER = "x-litellm-response-cost";
|
|
75170
76492
|
/** Response header naming the proxy deployment that served the call. */
|
|
75171
76493
|
const DEPLOYMENT_ID_HEADER = "x-litellm-model-id";
|
|
76494
|
+
/**
|
|
76495
|
+
* Response header in which the gateway reports the upstream model that
|
|
76496
|
+
* actually answered. Takes precedence over the body's `model`, which a proxy
|
|
76497
|
+
* may overwrite with the model-group name it was addressed by.
|
|
76498
|
+
*/
|
|
76499
|
+
const SERVED_MODEL_HEADER = "x-adaptic-served-model";
|
|
76500
|
+
/** Response header in which the gateway reports the upstream provider that answered. */
|
|
76501
|
+
const SERVED_PROVIDER_HEADER = "x-adaptic-served-provider";
|
|
75172
76502
|
/**
|
|
75173
76503
|
* Thrown when the gateway itself is unreachable, as opposed to a provider
|
|
75174
76504
|
* behind it failing.
|
|
@@ -75331,6 +76661,7 @@ function createGatewayTransport(config) {
|
|
|
75331
76661
|
throw new GatewayResponseError(response.status, await response.text());
|
|
75332
76662
|
}
|
|
75333
76663
|
const payload = (await response.json());
|
|
76664
|
+
const addressedAs = body.model;
|
|
75334
76665
|
const choices = payload.choices;
|
|
75335
76666
|
const message = choices?.[0]?.message;
|
|
75336
76667
|
// Usage is read before the content is interpreted. The provider billed for
|
|
@@ -75343,14 +76674,36 @@ function createGatewayTransport(config) {
|
|
|
75343
76674
|
tool_calls: Array.isArray(message?.tool_calls)
|
|
75344
76675
|
? message.tool_calls
|
|
75345
76676
|
: undefined,
|
|
75346
|
-
// The model
|
|
75347
|
-
//
|
|
75348
|
-
servedModel:
|
|
76677
|
+
// The model that answered, which a proxy-side fallback can make differ
|
|
76678
|
+
// from the leg's route model; unreported (or only the group echo) stays null.
|
|
76679
|
+
servedModel: servedModelOf(response.headers, payload.model, addressedAs),
|
|
75349
76680
|
servedDeploymentId: nonEmptyOrNull(response.headers?.get(DEPLOYMENT_ID_HEADER)),
|
|
76681
|
+
servedProvider: nonEmptyOrNull(response.headers?.get(SERVED_PROVIDER_HEADER)),
|
|
75350
76682
|
};
|
|
75351
76683
|
},
|
|
75352
76684
|
};
|
|
75353
76685
|
}
|
|
76686
|
+
/**
|
|
76687
|
+
* The model that answered, as far as the gateway says.
|
|
76688
|
+
*
|
|
76689
|
+
* The gateway's served-model header is authoritative when present. Otherwise
|
|
76690
|
+
* the body's `model` is used — unless it equals the name the leg was addressed
|
|
76691
|
+
* by, which is the proxy echoing its model group rather than naming the
|
|
76692
|
+
* upstream model, and would otherwise read as a confirmed same-model answer.
|
|
76693
|
+
*
|
|
76694
|
+
* @param headers The response headers.
|
|
76695
|
+
* @param bodyModel The body's `model` field.
|
|
76696
|
+
* @param addressedAs The gateway model name the leg was sent to.
|
|
76697
|
+
* @returns The served model, or null when the gateway did not say.
|
|
76698
|
+
*/
|
|
76699
|
+
function servedModelOf(headers, bodyModel, addressedAs) {
|
|
76700
|
+
const fromHeader = nonEmptyOrNull(headers?.get(SERVED_MODEL_HEADER));
|
|
76701
|
+
if (fromHeader !== null) {
|
|
76702
|
+
return fromHeader;
|
|
76703
|
+
}
|
|
76704
|
+
const fromBody = nonEmptyOrNull(bodyModel);
|
|
76705
|
+
return fromBody === null || fromBody === addressedAs ? null : fromBody;
|
|
76706
|
+
}
|
|
75354
76707
|
/**
|
|
75355
76708
|
* Compose the message array for a request.
|
|
75356
76709
|
*
|
|
@@ -75414,7 +76767,9 @@ function interpretContent(content, responseFormat, usage) {
|
|
|
75414
76767
|
* Every call gets, in order: alias resolution against the canonical route
|
|
75415
76768
|
* table, per-provider parameter normalisation, a hard per-leg timeout, a
|
|
75416
76769
|
* per-route circuit breaker, an ordered fallback chain ending at the closed
|
|
75417
|
-
* incumbent
|
|
76770
|
+
* incumbent — each leg first hedged and failed over on the SAME model (see
|
|
76771
|
+
* `hedge.ts`), and a different-model leg reached only when the caller's
|
|
76772
|
+
* cross-model policy allows it — and, where the caller supplies a validator, one schema-feedback
|
|
75418
76773
|
* retry ahead of the chain. The caller's `timeoutMs` is ONE deadline for the
|
|
75419
76774
|
* whole call: each leg runs for its route budget or for what remains of that
|
|
75420
76775
|
* deadline, whichever is shorter, so the chain is the single fallback owner and
|
|
@@ -75430,6 +76785,34 @@ const GATEWAY_BASE_URL_ENV = "LLM_GATEWAY_BASE_URL";
|
|
|
75430
76785
|
const DEFAULT_GATEWAY_KEY_ENV = "LLM_GATEWAY_API_KEY";
|
|
75431
76786
|
/** Process-wide breaker registry, so route health is shared across call sites. */
|
|
75432
76787
|
let breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker);
|
|
76788
|
+
/**
|
|
76789
|
+
* Build the process-wide latency tracker from the table's hedging defaults.
|
|
76790
|
+
*
|
|
76791
|
+
* @param now Clock.
|
|
76792
|
+
* @returns The tracker, or undefined when the table configures no hedging.
|
|
76793
|
+
*/
|
|
76794
|
+
function buildLatencyTracker(now) {
|
|
76795
|
+
const hedging = routeTable.defaults.hedging;
|
|
76796
|
+
if (hedging === undefined) {
|
|
76797
|
+
return undefined;
|
|
76798
|
+
}
|
|
76799
|
+
return new LegLatencyTracker({
|
|
76800
|
+
minSamples: hedging.min_samples,
|
|
76801
|
+
windowSize: hedging.window_size,
|
|
76802
|
+
sampleMaxAgeMs: hedging.sample_max_age_ms,
|
|
76803
|
+
promptTokenBuckets: hedging.prompt_token_buckets,
|
|
76804
|
+
}, now);
|
|
76805
|
+
}
|
|
76806
|
+
/**
|
|
76807
|
+
* Process-wide healthy-latency evidence, shared across call sites for the same
|
|
76808
|
+
* reason the breakers are: a model's health is one population however many
|
|
76809
|
+
* callers reach it.
|
|
76810
|
+
*/
|
|
76811
|
+
let latencyTracker = buildLatencyTracker();
|
|
76812
|
+
/** Same-model hedging policy from the table, or undefined when none is configured. */
|
|
76813
|
+
const hedgingPolicy = routeTable.defaults.hedging === undefined
|
|
76814
|
+
? undefined
|
|
76815
|
+
: sameModelPolicyFrom(routeTable.defaults.hedging);
|
|
75433
76816
|
/** Active runtime wiring. */
|
|
75434
76817
|
let config = {};
|
|
75435
76818
|
/** Lazily built transports, rebuilt whenever configuration changes. */
|
|
@@ -75451,6 +76834,28 @@ function configureLlmClient(next) {
|
|
|
75451
76834
|
gatewayTransport = null;
|
|
75452
76835
|
directTransport = null;
|
|
75453
76836
|
breakers = new CircuitBreakerRegistry(routeTable.defaults.circuit_breaker, next.now);
|
|
76837
|
+
latencyTracker = buildLatencyTracker(next.now);
|
|
76838
|
+
}
|
|
76839
|
+
/**
|
|
76840
|
+
* Inspect the healthy-latency evidence the hedging controls read.
|
|
76841
|
+
*
|
|
76842
|
+
* @returns The live tracker, or undefined when the table configures no hedging.
|
|
76843
|
+
*/
|
|
76844
|
+
function llmLatencyTracker() {
|
|
76845
|
+
return latencyTracker;
|
|
76846
|
+
}
|
|
76847
|
+
/**
|
|
76848
|
+
* Whether a duplicate attempt on the same provider may start.
|
|
76849
|
+
*
|
|
76850
|
+
* @param route The leg the duplicate would address.
|
|
76851
|
+
* @param reserveFraction Share of the provider's capacity kept free.
|
|
76852
|
+
* @returns Whether it may start.
|
|
76853
|
+
*/
|
|
76854
|
+
function admitDuplicate(route, reserveFraction) {
|
|
76855
|
+
if (config.duplicateAdmission !== undefined) {
|
|
76856
|
+
return config.duplicateAdmission(route, reserveFraction);
|
|
76857
|
+
}
|
|
76858
|
+
return hasDuplicateHeadroom(route.providerName, route.modelId, reserveFraction);
|
|
75454
76859
|
}
|
|
75455
76860
|
/**
|
|
75456
76861
|
* Inspect route health.
|
|
@@ -75523,10 +76928,11 @@ function directFor() {
|
|
|
75523
76928
|
* @param options The caller's options.
|
|
75524
76929
|
* @param responseFormat The requested response shape.
|
|
75525
76930
|
* @param transport The transport to carry every leg.
|
|
76931
|
+
* @param admitEquivalent Which same-model equivalents this transport can reach.
|
|
75526
76932
|
* @returns Prepared legs, in chain order.
|
|
75527
76933
|
*/
|
|
75528
|
-
function prepareLegs(chain, options, responseFormat, transport) {
|
|
75529
|
-
|
|
76934
|
+
function prepareLegs(chain, options, responseFormat, transport, admitEquivalent = () => true) {
|
|
76935
|
+
const prepare = (route) => {
|
|
75530
76936
|
try {
|
|
75531
76937
|
return {
|
|
75532
76938
|
route,
|
|
@@ -75540,7 +76946,11 @@ function prepareLegs(chain, options, responseFormat, transport) {
|
|
|
75540
76946
|
}
|
|
75541
76947
|
throw error;
|
|
75542
76948
|
}
|
|
75543
|
-
}
|
|
76949
|
+
};
|
|
76950
|
+
return chain.routes.map((route) => ({
|
|
76951
|
+
...prepare(route),
|
|
76952
|
+
equivalents: (route.equivalents ?? []).filter(admitEquivalent).map(prepare),
|
|
76953
|
+
}));
|
|
75544
76954
|
}
|
|
75545
76955
|
/**
|
|
75546
76956
|
* Call a language model by semantic alias.
|
|
@@ -75584,6 +76994,26 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75584
76994
|
}
|
|
75585
76995
|
const gateway = gatewayFor();
|
|
75586
76996
|
const attemptLog = [];
|
|
76997
|
+
// What every execution of this call shares, whichever transport carries it.
|
|
76998
|
+
// The configured model is the head of the full chain, so the degraded path —
|
|
76999
|
+
// whose legs are only the closed incumbents — still knows that its answer is
|
|
77000
|
+
// a different model from the one configured.
|
|
77001
|
+
const shared = {
|
|
77002
|
+
responseFormat,
|
|
77003
|
+
developerPrompt: options.developerPrompt,
|
|
77004
|
+
context: options.context,
|
|
77005
|
+
breakers,
|
|
77006
|
+
correlationId: options.correlationId,
|
|
77007
|
+
callerSignal: options.signal,
|
|
77008
|
+
deadlineAtMs,
|
|
77009
|
+
now: config.now,
|
|
77010
|
+
onAttempt: (record) => attemptLog.push(record),
|
|
77011
|
+
hedging: hedgingPolicy,
|
|
77012
|
+
latency: latencyTracker,
|
|
77013
|
+
admitDuplicate,
|
|
77014
|
+
crossModelPolicy: options.crossModelPolicy,
|
|
77015
|
+
configuredModelClass: modelClassOf(chain.routes[0]),
|
|
77016
|
+
};
|
|
75587
77017
|
/**
|
|
75588
77018
|
* Run the chain, falling back from the gateway to the degraded direct path
|
|
75589
77019
|
* only when the gateway itself is unreachable.
|
|
@@ -75596,17 +77026,9 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75596
77026
|
if (gateway !== null) {
|
|
75597
77027
|
try {
|
|
75598
77028
|
const outcome = await executeChain(options.alias, {
|
|
77029
|
+
...shared,
|
|
75599
77030
|
legs: prepareLegs(chain, options, responseFormat, gateway),
|
|
75600
77031
|
content: boundContent,
|
|
75601
|
-
responseFormat,
|
|
75602
|
-
developerPrompt: options.developerPrompt,
|
|
75603
|
-
context: options.context,
|
|
75604
|
-
breakers,
|
|
75605
|
-
correlationId: options.correlationId,
|
|
75606
|
-
callerSignal: options.signal,
|
|
75607
|
-
deadlineAtMs,
|
|
75608
|
-
now: config.now,
|
|
75609
|
-
onAttempt: (record) => attemptLog.push(record),
|
|
75610
77032
|
});
|
|
75611
77033
|
return { ...outcome, degraded: false };
|
|
75612
77034
|
}
|
|
@@ -75631,17 +77053,11 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75631
77053
|
});
|
|
75632
77054
|
}
|
|
75633
77055
|
const outcome = await executeChain(options.alias, {
|
|
75634
|
-
|
|
77056
|
+
...shared,
|
|
77057
|
+
// The direct transport serves closed vendors only, so only a closed
|
|
77058
|
+
// equivalent can be reached on the degraded path.
|
|
77059
|
+
legs: prepareLegs({ ...chain, routes: closedLegs }, options, responseFormat, direct, (route) => route.provider.tier === "closed"),
|
|
75635
77060
|
content: boundContent,
|
|
75636
|
-
responseFormat,
|
|
75637
|
-
developerPrompt: options.developerPrompt,
|
|
75638
|
-
context: options.context,
|
|
75639
|
-
breakers,
|
|
75640
|
-
correlationId: options.correlationId,
|
|
75641
|
-
callerSignal: options.signal,
|
|
75642
|
-
deadlineAtMs,
|
|
75643
|
-
now: config.now,
|
|
75644
|
-
onAttempt: (record) => attemptLog.push(record),
|
|
75645
77061
|
});
|
|
75646
77062
|
return { ...outcome, degraded: true };
|
|
75647
77063
|
};
|
|
@@ -75656,6 +77072,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75656
77072
|
attempts: attemptLog,
|
|
75657
77073
|
degraded: outcome.degraded,
|
|
75658
77074
|
totalUsage: outcome.totalUsage,
|
|
77075
|
+
modelClassRelation: outcome.modelClassRelation,
|
|
77076
|
+
hedged: outcome.hedged,
|
|
75659
77077
|
};
|
|
75660
77078
|
}
|
|
75661
77079
|
// A validator is only meaningful against a text prompt, because the retry has
|
|
@@ -75673,6 +77091,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75673
77091
|
servedBy: outcome.servedBy,
|
|
75674
77092
|
degraded: outcome.degraded,
|
|
75675
77093
|
totalUsage: outcome.totalUsage,
|
|
77094
|
+
modelClassRelation: outcome.modelClassRelation,
|
|
77095
|
+
hedged: outcome.hedged,
|
|
75676
77096
|
};
|
|
75677
77097
|
return outcome.response;
|
|
75678
77098
|
},
|
|
@@ -75690,6 +77110,8 @@ async function callLLMByAlias(content, responseFormat = "text", options) {
|
|
|
75690
77110
|
attempts: attemptLog,
|
|
75691
77111
|
degraded: routing.degraded,
|
|
75692
77112
|
totalUsage: validated.totalUsage,
|
|
77113
|
+
modelClassRelation: routing.modelClassRelation,
|
|
77114
|
+
hedged: routing.hedged,
|
|
75693
77115
|
};
|
|
75694
77116
|
}
|
|
75695
77117
|
/**
|
|
@@ -81151,13 +82573,17 @@ exports.DEFAULT_TRADING_POLICY = DEFAULT_TRADING_POLICY;
|
|
|
81151
82573
|
exports.DataFormatError = DataFormatError;
|
|
81152
82574
|
exports.DirectTransportRefusedError = DirectTransportRefusedError;
|
|
81153
82575
|
exports.DuplicateClientOrderIdError = DuplicateClientOrderIdError;
|
|
82576
|
+
exports.EQUIVALENT_SEPARATOR = EQUIVALENT_SEPARATOR;
|
|
81154
82577
|
exports.GatewayResponseError = GatewayResponseError;
|
|
81155
82578
|
exports.GatewayUnreachableError = GatewayUnreachableError;
|
|
81156
82579
|
exports.HttpClientError = HttpClientError;
|
|
81157
82580
|
exports.HttpServerError = HttpServerError;
|
|
81158
82581
|
exports.KEEP_ALIVE_DEFAULTS = KEEP_ALIVE_DEFAULTS;
|
|
82582
|
+
exports.LegLatencyTracker = LegLatencyTracker;
|
|
82583
|
+
exports.LlmDeadlineExceededError = LlmDeadlineExceededError;
|
|
81159
82584
|
exports.LlmResponseFormatError = LlmResponseFormatError;
|
|
81160
82585
|
exports.MARKET_DATA_API = MARKET_DATA_API;
|
|
82586
|
+
exports.MIN_CONVERTED_TRAIL_PERCENT = MIN_CONVERTED_TRAIL_PERCENT;
|
|
81161
82587
|
exports.MassiveAggregatesResponseSchema = MassiveAggregatesResponseSchema;
|
|
81162
82588
|
exports.MassiveApiError = MassiveApiError;
|
|
81163
82589
|
exports.MassiveDailyOpenCloseSchema = MassiveDailyOpenCloseSchema;
|
|
@@ -81180,6 +82606,8 @@ exports.RISK_FREE_RATE_TTL_MS = RISK_FREE_RATE_TTL_MS;
|
|
|
81180
82606
|
exports.RateGuardTimeoutError = RateGuardTimeoutError;
|
|
81181
82607
|
exports.RateLimitError = RateLimitError;
|
|
81182
82608
|
exports.RawMassivePriceDataSchema = RawMassivePriceDataSchema;
|
|
82609
|
+
exports.SERVED_MODEL_HEADER = SERVED_MODEL_HEADER;
|
|
82610
|
+
exports.SERVED_PROVIDER_HEADER = SERVED_PROVIDER_HEADER;
|
|
81183
82611
|
exports.SchemaRetryExhaustedError = SchemaRetryExhaustedError;
|
|
81184
82612
|
exports.StampedeProtectedCache = StampedeProtectedCache;
|
|
81185
82613
|
exports.StreamProviderError = StreamProviderError;
|
|
@@ -81189,6 +82617,7 @@ exports.TimeoutError = TimeoutError;
|
|
|
81189
82617
|
exports.TokenBucketRateLimiter = TokenBucketRateLimiter;
|
|
81190
82618
|
exports.ToolChoiceIgnoredError = ToolChoiceIgnoredError;
|
|
81191
82619
|
exports.TradeError = TradeError;
|
|
82620
|
+
exports.TrailUnitConversionRefusedError = TrailUnitConversionRefusedError;
|
|
81192
82621
|
exports.TrailingStopValidationError = TrailingStopValidationError;
|
|
81193
82622
|
exports.USDC_PAIRS = USDC_PAIRS;
|
|
81194
82623
|
exports.USDT_PAIRS = USDT_PAIRS;
|
|
@@ -81274,6 +82703,7 @@ exports.createVerticalSpread = createVerticalSpread$1;
|
|
|
81274
82703
|
exports.createVerticalSpreadAdvanced = createVerticalSpread;
|
|
81275
82704
|
exports.enrichAlpacaError = enrichAlpacaError;
|
|
81276
82705
|
exports.entryWithPercentStopLoss = entryWithPercentStopLoss;
|
|
82706
|
+
exports.estimatePromptTokens = estimatePromptTokens;
|
|
81277
82707
|
exports.exerciseOption = exerciseOption;
|
|
81278
82708
|
exports.extractAlpacaBrokerError = extractAlpacaBrokerError;
|
|
81279
82709
|
exports.extractGreeks = extractGreeks;
|
|
@@ -81376,6 +82806,7 @@ exports.groupOrdersByStatus = groupOrdersByStatus;
|
|
|
81376
82806
|
exports.groupOrdersBySymbol = groupOrdersBySymbol;
|
|
81377
82807
|
exports.guardSnapshots = guardSnapshots;
|
|
81378
82808
|
exports.hasActiveTrailingStop = hasActiveTrailingStop;
|
|
82809
|
+
exports.hasDuplicateHeadroom = hasDuplicateHeadroom;
|
|
81379
82810
|
exports.hasOptionLiquidity = hasGoodLiquidity;
|
|
81380
82811
|
exports.hasStockLiquidity = hasGoodLiquidity$1;
|
|
81381
82812
|
exports.hasSufficientVolume = hasSufficientVolume;
|
|
@@ -81394,6 +82825,7 @@ exports.isOrderFilled = isOrderFilled;
|
|
|
81394
82825
|
exports.isOrderOpen = isOrderOpen;
|
|
81395
82826
|
exports.isOrderTerminalStatus = isOrderTerminal$1;
|
|
81396
82827
|
exports.isPendingCancelRejection = isPendingCancelRejection;
|
|
82828
|
+
exports.isSameReportedModel = isSameReportedModel;
|
|
81397
82829
|
exports.isSupportedCryptoPair = isSupportedCryptoPair;
|
|
81398
82830
|
exports.isTransientNetworkError = isTransientNetworkError;
|
|
81399
82831
|
exports.legBudgetMs = legBudgetMs;
|
|
@@ -81404,6 +82836,9 @@ exports.limitsInventory = limitsInventory;
|
|
|
81404
82836
|
exports.listAliases = listAliases;
|
|
81405
82837
|
exports.llmAliases = llmAliases;
|
|
81406
82838
|
exports.llmBreakers = llmBreakers;
|
|
82839
|
+
exports.llmLatencyTracker = llmLatencyTracker;
|
|
82840
|
+
exports.modelClassOf = modelClassOf;
|
|
82841
|
+
exports.modelClassRelationOf = modelClassRelationOf;
|
|
81407
82842
|
exports.normaliseAnthropicStream = normaliseAnthropicStream;
|
|
81408
82843
|
exports.normaliseOpenAiStream = normaliseOpenAiStream;
|
|
81409
82844
|
exports.normaliseParams = normaliseParams;
|
|
@@ -81418,11 +82853,13 @@ exports.parseOCCSymbol = parseOCCSymbol;
|
|
|
81418
82853
|
exports.protectLongPosition = protectLongPosition;
|
|
81419
82854
|
exports.protectShortPosition = protectShortPosition;
|
|
81420
82855
|
exports.rateLimiters = rateLimiters$1;
|
|
82856
|
+
exports.readTrailUnit = readTrailUnit;
|
|
81421
82857
|
exports.resetLogger = resetLogger;
|
|
81422
82858
|
exports.resetProviderGuards = resetProviderGuards;
|
|
81423
82859
|
exports.resetRiskFreeRateCache = resetRiskFreeRateCache;
|
|
81424
82860
|
exports.resolveChain = resolveChain;
|
|
81425
82861
|
exports.resolveDefaultDirectCaller = resolveDefaultDirectCaller;
|
|
82862
|
+
exports.resolveReplaceTrail = resolveReplaceTrail;
|
|
81426
82863
|
exports.risk = riskNs;
|
|
81427
82864
|
exports.rollOptionPosition = rollOptionPosition;
|
|
81428
82865
|
exports.roundPriceForAlpaca = roundPriceForAlpaca$3;
|
|
@@ -81437,12 +82874,14 @@ exports.sellAllCrypto = sellAllCrypto;
|
|
|
81437
82874
|
exports.sellCryptoNotional = sellCryptoNotional;
|
|
81438
82875
|
exports.sellToClose = sellToClose;
|
|
81439
82876
|
exports.sellToOpen = sellToOpen;
|
|
82877
|
+
exports.servedModelOf = servedModelOf;
|
|
81440
82878
|
exports.setLogger = setLogger;
|
|
81441
82879
|
exports.setRiskFreeRate = setRiskFreeRate;
|
|
81442
82880
|
exports.shortWithStopLoss = shortWithStopLoss;
|
|
81443
82881
|
exports.sortOrdersByDate = sortOrdersByDate;
|
|
81444
82882
|
exports.strategy = strategyNs;
|
|
81445
82883
|
exports.sumUsage = sumUsage;
|
|
82884
|
+
exports.tailLatencyViolations = tailLatencyViolations;
|
|
81446
82885
|
exports.tradingPolicy = index;
|
|
81447
82886
|
exports.trailingStops = trailingStops;
|
|
81448
82887
|
exports.unavailableStatistic = unavailableStatistic;
|